{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/ran/4","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":4,"pages_in_order":12,"rows_per_page":100,"rows":[301,400],"of":1165,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2/papers/ran/1","prev":"/task/reinforcement-learning-2/papers/ran/3","next":"/task/reinforcement-learning-2/papers/ran/5","papers":[{"url":"/paper/reasoning-with-latent-diffusion-in-offline","slug":"reasoning-with-latent-diffusion-in-offline","title":"Reasoning with Latent Diffusion in Offline Reinforcement Learning","date":"2023-09-12","arxiv_id":"2309.06599","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reasoning-with-latent-diffusion-in-offline#ran","syntology_url":"https://syntology.ai/paper/2309.06599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06599"}},"official":{"repos":["ldcq/ldcq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/subwords-as-skills-tokenization-for-sparse","slug":"subwords-as-skills-tokenization-for-sparse","title":"Subwords as Skills: Tokenization for Sparse-Reward Reinforcement Learning","date":"2023-09-08","arxiv_id":"2309.04459","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/subwords-as-skills-tokenization-for-sparse#ran","syntology_url":"https://syntology.ai/paper/2309.04459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04459"}},"official":{"repos":["dyunis/subwords_as_skills"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-reinforcement-learning-training","slug":"improving-reinforcement-learning-training","title":"Improving Generalization in Reinforcement Learning Training Regimes for Social Robot Navigation","date":"2023-08-29","arxiv_id":"2308.14947","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-reinforcement-learning-training#ran","syntology_url":"https://syntology.ai/paper/2308.14947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14947"}},"official":{"repos":["raise-lab/soc-nav-training"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-reward-modulation-for-pretraining","slug":"language-reward-modulation-for-pretraining","title":"Language Reward Modulation for Pretraining Reinforcement Learning","date":"2023-08-23","arxiv_id":"2308.12270","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-reward-modulation-for-pretraining#ran","syntology_url":"https://syntology.ai/paper/2308.12270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12270"}},"official":{"repos":["ademiadeniji/lamp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discrete-prompt-compression-with","slug":"discrete-prompt-compression-with","title":"Discrete Prompt Compression with Reinforcement Learning","date":"2023-08-17","arxiv_id":"2308.08758","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-prompt-compression-with#ran","syntology_url":"https://syntology.ai/paper/2308.08758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08758"}},"official":{"repos":["nenomigami/promptcompressor"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-identify-critical-states-for","slug":"learning-to-identify-critical-states-for","title":"Learning to Identify Critical States for Reinforcement Learning from Videos","date":"2023-08-15","arxiv_id":"2308.07795","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-identify-critical-states-for#ran","syntology_url":"https://syntology.ai/paper/2308.07795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07795"}},"official":{"repos":["ai-initiative-kaust/videorlcs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alphastar-unplugged-large-scale-offline","slug":"alphastar-unplugged-large-scale-offline","title":"AlphaStar Unplugged: Large-Scale Offline Reinforcement Learning","date":"2023-08-07","arxiv_id":"2308.03526","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alphastar-unplugged-large-scale-offline#ran","syntology_url":"https://syntology.ai/paper/2308.03526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03526"}},"official":null}},{"url":"/paper/esrl-efficient-sampling-based-reinforcement","slug":"esrl-efficient-sampling-based-reinforcement","title":"ESRL: Efficient Sampling-based Reinforcement Learning for Sequence Generation","date":"2023-08-04","arxiv_id":"2308.02223","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/esrl-efficient-sampling-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2308.02223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02223"}},"official":{"repos":["wangclnlp/DeepSpeed-Chat-Extension"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-non","slug":"reinforcement-learning-based-non","title":"Reinforcement Learning-based Non-Autoregressive Solver for Traveling Salesman Problems","date":"2023-08-01","arxiv_id":"2308.00560","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-based-non#ran","syntology_url":"https://syntology.ai/paper/2308.00560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00560"}},"official":{"repos":["xybfight/nar4tsp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/value-informed-skill-chaining-for-policy","slug":"value-informed-skill-chaining-for-policy","title":"Value-Informed Skill Chaining for Policy Learning of Long-Horizon Tasks with Surgical Robot","date":"2023-07-31","arxiv_id":"2307.16503","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/value-informed-skill-chaining-for-policy#ran","syntology_url":"https://syntology.ai/paper/2307.16503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16503"}},"official":{"repos":["med-air/viskill"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-multi-agent-reinforcement-learning-3","slug":"robust-multi-agent-reinforcement-learning-3","title":"Robust Multi-Agent Reinforcement Learning with State Uncertainty","date":"2023-07-30","arxiv_id":"2307.16212","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/robust-multi-agent-reinforcement-learning-3#ran","syntology_url":"https://syntology.ai/paper/2307.16212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16212"}},"official":{"repos":["sihongho/robust_marl_with_state_uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/submodular-reinforcement-learning","slug":"submodular-reinforcement-learning","title":"Submodular Reinforcement Learning","date":"2023-07-25","arxiv_id":"2307.13372","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/submodular-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.13372","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.13372"}},"official":{"repos":["manish-pra/non-additive-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rlcd-reinforcement-learning-from-contrast","slug":"rlcd-reinforcement-learning-from-contrast","title":"RLCD: Reinforcement Learning from Contrastive Distillation for Language Model Alignment","date":"2023-07-24","arxiv_id":"2307.12950","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rlcd-reinforcement-learning-from-contrast#ran","syntology_url":"https://syntology.ai/paper/2307.12950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12950"}},"official":{"repos":["facebookresearch/rlcd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/parallel-q-learning-scaling-off-policy","slug":"parallel-q-learning-scaling-off-policy","title":"Parallel $Q$-Learning: Scaling Off-policy Reinforcement Learning under Massively Parallel Simulation","date":"2023-07-24","arxiv_id":"2307.12983","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallel-q-learning-scaling-off-policy#ran","syntology_url":"https://syntology.ai/paper/2307.12983","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12983"}},"official":null}},{"url":"/paper/balancing-exploration-and-exploitation-in","slug":"balancing-exploration-and-exploitation-in","title":"Balancing Exploration and Exploitation in Hierarchical Reinforcement Learning via Latent Landmark Graphs","date":"2023-07-22","arxiv_id":"2307.12063","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/balancing-exploration-and-exploitation-in#ran","syntology_url":"https://syntology.ai/paper/2307.12063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12063"}},"official":{"repos":["papercode2022/hill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emergence-of-adaptive-circadian-rhythms-in","slug":"emergence-of-adaptive-circadian-rhythms-in","title":"Emergence of Adaptive Circadian Rhythms in Deep Reinforcement Learning","date":"2023-07-22","arxiv_id":"2307.12143","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergence-of-adaptive-circadian-rhythms-in#ran","syntology_url":"https://syntology.ai/paper/2307.12143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12143"}},"official":{"repos":["aqeel13932/mn_project"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-offline-reinforcement-learning-2","slug":"model-based-offline-reinforcement-learning-2","title":"Model-based Offline Reinforcement Learning with Count-based Conservatism","date":"2023-07-21","arxiv_id":"2307.11352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/model-based-offline-reinforcement-learning-2#ran","syntology_url":"https://syntology.ai/paper/2307.11352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11352"}},"official":{"repos":["oh-lab/count-morl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/natural-actor-critic-for-robust-reinforcement","slug":"natural-actor-critic-for-robust-reinforcement","title":"Natural Actor-Critic for Robust Reinforcement Learning with Function Approximation","date":"2023-07-17","arxiv_id":"2307.08875","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/natural-actor-critic-for-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2307.08875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08875"}},"official":{"repos":["tliu1997/rnac"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-vigen-a-reinforcement-learning-benchmark-1","slug":"rl-vigen-a-reinforcement-learning-benchmark-1","title":"RL-ViGen: A Reinforcement Learning Benchmark for Visual Generalization","date":"2023-07-15","arxiv_id":"2307.10224","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rl-vigen-a-reinforcement-learning-benchmark-1#ran","syntology_url":"https://syntology.ai/paper/2307.10224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10224"}},"official":{"repos":["gemcollector/rl-vigen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-dreamerv3-safe-reinforcement-learning","slug":"safe-dreamerv3-safe-reinforcement-learning","title":"SafeDreamer: Safe Reinforcement Learning with World Models","date":"2023-07-14","arxiv_id":"2307.07176","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-dreamerv3-safe-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.07176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07176"}},"official":{"repos":["pku-alignment/safedreamer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rltf-reinforcement-learning-from-unit-test","slug":"rltf-reinforcement-learning-from-unit-test","title":"RLTF: Reinforcement Learning from Unit Test Feedback","date":"2023-07-10","arxiv_id":"2307.04349","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rltf-reinforcement-learning-from-unit-test#ran","syntology_url":"https://syntology.ai/paper/2307.04349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04349"}},"official":{"repos":["zyq-scut/rltf"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/alleviating-matthew-effect-of-offline","slug":"alleviating-matthew-effect-of-offline","title":"Alleviating Matthew Effect of Offline Reinforcement Learning in Interactive Recommendation","date":"2023-07-10","arxiv_id":"2307.04571","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alleviating-matthew-effect-of-offline#ran","syntology_url":"https://syntology.ai/paper/2307.04571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04571"}},"official":{"repos":["chongminggao/dorl-codes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-symbolic-rules-over-abstract-meaning","slug":"learning-symbolic-rules-over-abstract-meaning","title":"Learning Symbolic Rules over Abstract Meaning Representations for Textual Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02689","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-symbolic-rules-over-abstract-meaning#ran","syntology_url":"https://syntology.ai/paper/2307.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02689"}},"official":{"repos":["ibm/loa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-reinforcement-learning-for-online","slug":"end-to-end-reinforcement-learning-for-online","title":"Learning Coverage Paths in Unknown Environments with Deep Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.16978","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-reinforcement-learning-for-online#ran","syntology_url":"https://syntology.ai/paper/2306.16978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16978"}},"official":{"repos":["arvijj/rl-cpp"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/rl4co-an-extensive-reinforcement-learning-for","slug":"rl4co-an-extensive-reinforcement-learning-for","title":"RL4CO: an Extensive Reinforcement Learning for Combinatorial Optimization Benchmark","date":"2023-06-29","arxiv_id":"2306.17100","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rl4co-an-extensive-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17100"}},"official":{"repos":["ai4co/rl4co","pytorch/rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/beyond-ood-state-actions-supported-cross","slug":"beyond-ood-state-actions-supported-cross","title":"Beyond OOD State Actions: Supported Cross-Domain Offline Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.12755","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-ood-state-actions-supported-cross#ran","syntology_url":"https://syntology.ai/paper/2306.12755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12755"}},"official":{"repos":["thuml/SPOT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taco-temporal-latent-action-driven","slug":"taco-temporal-latent-action-driven","title":"TACO: Temporal Latent Action-Driven Contrastive Loss for Visual Reinforcement Learning","date":"2023-06-22","arxiv_id":"2306.13229","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/taco-temporal-latent-action-driven#ran","syntology_url":"https://syntology.ai/paper/2306.13229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13229"}},"official":{"repos":["frankzheng2022/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/jumanji-a-diverse-suite-of-scalable","slug":"jumanji-a-diverse-suite-of-scalable","title":"Jumanji: a Diverse Suite of Scalable Reinforcement Learning Environments in JAX","date":"2023-06-16","arxiv_id":"2306.09884","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jumanji-a-diverse-suite-of-scalable#ran","syntology_url":"https://syntology.ai/paper/2306.09884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09884"}},"official":{"repos":["instadeepai/jumanji"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","slug":"datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/datasets-and-benchmarks-for-offline-safe#ran","syntology_url":"https://syntology.ai/paper/2306.09303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09303"}},"official":{"repos":["liuzuxin/dsrl","liuzuxin/fsrl","liuzuxin/osrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-memory-decision-transformer","slug":"recurrent-memory-decision-transformer","title":"Recurrent Action Transformer with Memory","date":"2023-06-15","arxiv_id":"2306.09459","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-memory-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2306.09459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09459"}},"official":{"repos":["airi-institute/rate"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mediated-multi-agent-reinforcement-learning","slug":"mediated-multi-agent-reinforcement-learning","title":"Mediated Multi-Agent Reinforcement Learning","date":"2023-06-14","arxiv_id":"2306.08419","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mediated-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2306.08419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08419"}},"official":{"repos":["dimonenka/mediatedmarl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ocatari-object-centric-atari-2600","slug":"ocatari-object-centric-atari-2600","title":"OCAtari: Object-Centric Atari 2600 Reinforcement Learning Environments","date":"2023-06-14","arxiv_id":"2306.08649","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocatari-object-centric-atari-2600#ran","syntology_url":"https://syntology.ai/paper/2306.08649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08649"}},"official":{"repos":["k4ntz/oc_atari"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/katakomba-tools-and-benchmarks-for-data","slug":"katakomba-tools-and-benchmarks-for-data","title":"Katakomba: Tools and Benchmarks for Data-Driven NetHack","date":"2023-06-14","arxiv_id":"2306.08772","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/katakomba-tools-and-benchmarks-for-data#ran","syntology_url":"https://syntology.ai/paper/2306.08772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08772"}},"official":{"repos":["corl-team/katakomba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-versatile-multi-agent-reinforcement","slug":"a-versatile-multi-agent-reinforcement","title":"A Versatile Multi-Agent Reinforcement Learning Benchmark for Inventory Management","date":"2023-06-13","arxiv_id":"2306.07542","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-versatile-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.07542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07542"}},"official":{"repos":["victoryxl/replenishmentenv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-regularization-with-dataset-constraint","slug":"policy-regularization-with-dataset-constraint","title":"Policy Regularization with Dataset Constraint for Offline Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06569","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/policy-regularization-with-dataset-constraint#ran","syntology_url":"https://syntology.ai/paper/2306.06569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06569"}},"official":{"repos":["lamda-rl/prdc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/generalizable-wireless-navigation-through","slug":"generalizable-wireless-navigation-through","title":"Digital Twin-Enhanced Wireless Indoor Navigation: Achieving Efficient Environment Sensing with Zero-Shot Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06766","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-wireless-navigation-through#ran","syntology_url":"https://syntology.ai/paper/2306.06766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06766"}},"official":{"repos":["panshark/pirl-win"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-efficacy-of-3d-point-cloud","slug":"on-the-efficacy-of-3d-point-cloud","title":"On the Efficacy of 3D Point Cloud Reinforcement Learning","date":"2023-06-11","arxiv_id":"2306.06799","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-the-efficacy-of-3d-point-cloud#ran","syntology_url":"https://syntology.ai/paper/2306.06799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06799"}},"official":{"repos":["lz1oceani/pointcloud_rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-reinforcement-learning-with","slug":"explaining-reinforcement-learning-with","title":"Explaining Reinforcement Learning with Shapley Values","date":"2023-06-09","arxiv_id":"2306.05810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explaining-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2306.05810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05810"}},"official":{"repos":["bath-reinforcement-learning-lab/sverl_icml_2023"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-semi-parametric-1","slug":"large-language-models-are-semi-parametric-1","title":"Large Language Models Are Semi-Parametric Reinforcement Learning Agents","date":"2023-06-09","arxiv_id":"2306.07929","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-semi-parametric-1#ran","syntology_url":"https://syntology.ai/paper/2306.07929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07929"}},"official":{"repos":["opendfm/rememberer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flipping-coins-to-estimate-pseudocounts-for","slug":"flipping-coins-to-estimate-pseudocounts-for","title":"Flipping Coins to Estimate Pseudocounts for Exploration in Reinforcement Learning","date":"2023-06-05","arxiv_id":"2306.03186","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/flipping-coins-to-estimate-pseudocounts-for#ran","syntology_url":"https://syntology.ai/paper/2306.03186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03186"}},"official":{"repos":["samlobel/cfn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-aware-reward-shaping-of-reinforcement","slug":"risk-aware-reward-shaping-of-reinforcement","title":"Risk-Aware Reward Shaping of Reinforcement Learning Agents for Autonomous Driving","date":"2023-06-05","arxiv_id":"2306.03220","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/risk-aware-reward-shaping-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.03220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03220"}},"official":{"repos":["zhang-zengjie/code_2023_iecon_shaping_wu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/ma2cl-masked-attentive-contrastive-learning","slug":"ma2cl-masked-attentive-contrastive-learning","title":"MA2CL:Masked Attentive Contrastive Learning for Multi-Agent Reinforcement Learning","date":"2023-06-03","arxiv_id":"2306.02006","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ma2cl-masked-attentive-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2306.02006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02006"}},"official":{"repos":["ustchlsong/ma2cl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameters-in-reinforcement-learning-and","slug":"hyperparameters-in-reinforcement-learning-and","title":"Hyperparameters in Reinforcement Learning and How To Tune Them","date":"2023-06-02","arxiv_id":"2306.01324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hyperparameters-in-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/2306.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01324"}},"official":{"repos":["facebookresearch/how-to-autorl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tackling-unbounded-state-spaces-in-continuing","slug":"tackling-unbounded-state-spaces-in-continuing","title":"Learning to Stabilize Online Reinforcement Learning in Unbounded State Spaces","date":"2023-06-02","arxiv_id":"2306.01896","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":1,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tackling-unbounded-state-spaces-in-continuing#ran","syntology_url":"https://syntology.ai/paper/2306.01896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01896"}},"official":{"repos":["badger-rl/stop"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/identifiability-and-generalizability-in","slug":"identifiability-and-generalizability-in","title":"Identifiability and Generalizability in Constrained Inverse Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00629","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/identifiability-and-generalizability-in#ran","syntology_url":"https://syntology.ai/paper/2306.00629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00629"}},"official":{"repos":["andrschl/cirl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/normalization-enhances-generalization-in","slug":"normalization-enhances-generalization-in","title":"Normalization Enhances Generalization in Visual Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00656","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/normalization-enhances-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2306.00656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00656"}},"official":{"repos":["lilucse/Normalization-Enhances-Generalization-in-Visual-Reinforcement-Learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/active-vision-reinforcement-learning-under","slug":"active-vision-reinforcement-learning-under","title":"Active Vision Reinforcement Learning under Limited Visual Observability","date":"2023-06-01","arxiv_id":"2306.00975","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":4,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/active-vision-reinforcement-learning-under#ran","syntology_url":"https://syntology.ai/paper/2306.00975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00975"}},"official":{"repos":["elicassion/sugarl","elicassion/active-gym"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-meta-reinforcement-learning-with-in","slug":"offline-meta-reinforcement-learning-with-in","title":"Offline Meta Reinforcement Learning with In-Distribution Online Adaptation","date":"2023-05-31","arxiv_id":"2305.19529","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with-in#ran","syntology_url":"https://syntology.ai/paper/2305.19529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19529"}},"official":{"repos":["nagisazj/idaq_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-diffusion-policies-for-offline-1","slug":"efficient-diffusion-policies-for-offline-1","title":"Efficient Diffusion Policies for Offline Reinforcement Learning","date":"2023-05-31","arxiv_id":"2305.20081","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-diffusion-policies-for-offline-1#ran","syntology_url":"https://syntology.ai/paper/2305.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20081"}},"official":{"repos":["sail-sg/edp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rosarl-reward-only-safe-reinforcement","slug":"rosarl-reward-only-safe-reinforcement","title":"ROSARL: Reward-Only Safe Reinforcement Learning","date":"2023-05-31","arxiv_id":"2306.00035","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rosarl-reward-only-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2306.00035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00035"}},"official":{"repos":["geraudnt/rosarl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/subequivariant-graph-reinforcement-learning","slug":"subequivariant-graph-reinforcement-learning","title":"Subequivariant Graph Reinforcement Learning in 3D Environments","date":"2023-05-30","arxiv_id":"2305.18951","repositories_listed":1,"syntology":{"n":18,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/subequivariant-graph-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2305.18951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18951"}},"official":{"repos":["alpc91/sgrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/provable-and-practical-efficient-exploration","slug":"provable-and-practical-efficient-exploration","title":"Provable and Practical: Efficient Exploration in Reinforcement Learning via Langevin Monte Carlo","date":"2023-05-29","arxiv_id":"2305.18246","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provable-and-practical-efficient-exploration#ran","syntology_url":"https://syntology.ai/paper/2305.18246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18246"}},"official":{"repos":["hmishfaq/lmc-lsvi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-optimization-your-language","slug":"direct-preference-optimization-your-language","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","date":"2023-05-29","arxiv_id":"2305.18290","repositories_listed":29,"syntology":{"n":31,"n_ran":25,"n_constructed":1,"n_ran_checked":24,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":24,"n_pointer_only":2,"phrase":"25 ran (of which 1 constructed an object rather than computing a result; 24 with no instrument failure: 0 honoured, 0 violated, 24 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/direct-preference-optimization-your-language#ran","syntology_url":"https://syntology.ai/paper/2305.18290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18290"}},"official":null}},{"url":"/paper/is-centralized-training-with-decentralized","slug":"is-centralized-training-with-decentralized","title":"Is Centralized Training with Decentralized Execution Framework Centralized Enough for MARL?","date":"2023-05-27","arxiv_id":"2305.17352","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-centralized-training-with-decentralized#ran","syntology_url":"https://syntology.ai/paper/2305.17352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17352"}},"official":{"repos":["zyh1999/cadp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/query-policy-misalignment-in-preference-based","slug":"query-policy-misalignment-in-preference-based","title":"Query-Policy Misalignment in Preference-Based Reinforcement Learning","date":"2023-05-27","arxiv_id":"2305.17400","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/query-policy-misalignment-in-preference-based#ran","syntology_url":"https://syntology.ai/paper/2305.17400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17400"}},"official":{"repos":["huxiao09/qpa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-reward-offline-preference-guided","slug":"beyond-reward-offline-preference-guided","title":"Beyond Reward: Offline Preference-guided Policy Optimization","date":"2023-05-25","arxiv_id":"2305.16217","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-reward-offline-preference-guided#ran","syntology_url":"https://syntology.ai/paper/2305.16217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16217"}},"official":{"repos":["bkkgbkjb/oppo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coherent-soft-imitation-learning","slug":"coherent-soft-imitation-learning","title":"Coherent Soft Imitation Learning","date":"2023-05-25","arxiv_id":"2305.16498","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coherent-soft-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2305.16498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16498"}},"official":{"repos":["google-deepmind/csil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-language-models-with-advantage","slug":"improving-language-models-with-advantage","title":"Leftover Lunch: Advantage-based Offline Reinforcement Learning for Language Models","date":"2023-05-24","arxiv_id":"2305.14718","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-language-models-with-advantage#ran","syntology_url":"https://syntology.ai/paper/2305.14718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14718"}},"official":{"repos":["abaheti95/lol-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-policy-adapters-ipa-tailoring","slug":"inference-time-policy-adapters-ipa-tailoring","title":"Inference-Time Policy Adapters (IPA): Tailoring Extreme-Scale LMs without Fine-tuning","date":"2023-05-24","arxiv_id":"2305.15065","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":3,"n_ran_checked":5,"n_instrument":4,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/inference-time-policy-adapters-ipa-tailoring#ran","syntology_url":"https://syntology.ai/paper/2305.15065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15065"}},"official":{"repos":["gximinglu/ipa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/guard-a-safe-reinforcement-learning-benchmark","slug":"guard-a-safe-reinforcement-learning-benchmark","title":"GUARD: A Safe Reinforcement Learning Benchmark","date":"2023-05-23","arxiv_id":"2305.13681","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guard-a-safe-reinforcement-learning-benchmark#ran","syntology_url":"https://syntology.ai/paper/2305.13681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13681"}},"official":{"repos":["intelligent-control-lab/guard"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conditional-mutual-information-for-1","slug":"conditional-mutual-information-for-1","title":"Conditional Mutual Information for Disentangled Representations in Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-mutual-information-for-1#ran","syntology_url":"https://syntology.ai/paper/2305.14133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14133"}},"official":{"repos":["uoe-agents/cmid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-prediction-models-as-rewards-for","slug":"video-prediction-models-as-rewards-for","title":"Video Prediction Models as Rewards for Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14343","repositories_listed":3,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/video-prediction-models-as-rewards-for#ran","syntology_url":"https://syntology.ai/paper/2305.14343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14343"}},"official":null}},{"url":"/paper/2305-14550","slug":"2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","arxiv_id":"2305.14550","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2305-14550#ran","syntology_url":"https://syntology.ai/paper/2305.14550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14550"}},"official":{"repos":["prajjwal1/rl_paradigm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-representation-via-diffusion","slug":"policy-representation-via-diffusion","title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13122","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-representation-via-diffusion#ran","syntology_url":"https://syntology.ai/paper/2305.13122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13122"}},"official":{"repos":["bellmantimehut/dipo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/training-diffusion-models-with-reinforcement","slug":"training-diffusion-models-with-reinforcement","title":"Training Diffusion Models with Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13301","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/training-diffusion-models-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.13301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13301"}},"official":{"repos":["kvablack/ddpo-pytorch"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-diverse-risk-preferences-in","slug":"learning-diverse-risk-preferences-in","title":"Learning Diverse Risk Preferences in Population-based Self-play","date":"2023-05-19","arxiv_id":"2305.11476","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-diverse-risk-preferences-in#ran","syntology_url":"https://syntology.ai/paper/2305.11476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11476"}},"official":{"repos":["jackory/rpbt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/demonstration-free-autonomous-reinforcement","slug":"demonstration-free-autonomous-reinforcement","title":"Demonstration-free Autonomous Reinforcement Learning via Implicit and Bidirectional Curriculum","date":"2023-05-17","arxiv_id":"2305.09943","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/demonstration-free-autonomous-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.09943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09943"}},"official":{"repos":["snu-larr/ibc_official"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-minimalist-approach-to-offline","slug":"revisiting-the-minimalist-approach-to-offline","title":"Revisiting the Minimalist Approach to Offline Reinforcement Learning","date":"2023-05-16","arxiv_id":"2305.09836","repositories_listed":3,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-minimalist-approach-to-offline#ran","syntology_url":"https://syntology.ai/paper/2305.09836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09836"}},"official":{"repos":["dt6a/rebrac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/rl4f-generating-natural-language-feedback","slug":"rl4f-generating-natural-language-feedback","title":"RL4F: Generating Natural Language Feedback with Reinforcement Learning for Repairing Model Outputs","date":"2023-05-15","arxiv_id":"2305.08844","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rl4f-generating-natural-language-feedback#ran","syntology_url":"https://syntology.ai/paper/2305.08844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.08844"}},"official":{"repos":["feyzaakyurek/rl4f"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/towards-generalizable-reinforcement-learning","slug":"towards-generalizable-reinforcement-learning","title":"Towards Generalizable Reinforcement Learning for Trade Execution","date":"2023-05-12","arxiv_id":"2307.11685","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/towards-generalizable-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.11685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11685"}},"official":null}},{"url":"/paper/local-optimization-achieves-global-optimality","slug":"local-optimization-achieves-global-optimality","title":"Local Optimization Achieves Global Optimality in Multi-Agent Reinforcement Learning","date":"2023-05-08","arxiv_id":"2305.04819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/local-optimization-achieves-global-optimality#ran","syntology_url":"https://syntology.ai/paper/2305.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04819"}},"official":{"repos":["zhaoyl18/ratio_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-factored-action-spaces-for","slug":"leveraging-factored-action-spaces-for","title":"Leveraging Factored Action Spaces for Efficient Offline Reinforcement Learning in Healthcare","date":"2023-05-02","arxiv_id":"2305.01738","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-factored-action-spaces-for#ran","syntology_url":"https://syntology.ai/paper/2305.01738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.01738"}},"official":{"repos":["mld3/offlinerl_factoredactions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-achievement-structure-for-structured","slug":"learning-achievement-structure-for-structured","title":"Learning Achievement Structure for Structured Exploration in Domains with Sparse Reward","date":"2023-04-30","arxiv_id":"2305.00508","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-achievement-structure-for-structured#ran","syntology_url":"https://syntology.ai/paper/2305.00508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.00508"}},"official":{"repos":["pairlab/iclr-23-sea"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/quantum-natural-policy-gradients-towards","slug":"quantum-natural-policy-gradients-towards","title":"Quantum Natural Policy Gradients: Towards Sample-Efficient Reinforcement Learning","date":"2023-04-26","arxiv_id":"2304.13571","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantum-natural-policy-gradients-towards#ran","syntology_url":"https://syntology.ai/paper/2304.13571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13571"}},"official":{"repos":["nicomeyer96/quantum-natural-policy-gradients"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-datasets-and-market-environments-for","slug":"dynamic-datasets-and-market-environments-for","title":"Dynamic Datasets and Market Environments for Financial Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.13174","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-datasets-and-market-environments-for#ran","syntology_url":"https://syntology.ai/paper/2304.13174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13174"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-deep-reinforcement-learning","slug":"efficient-deep-reinforcement-learning","title":"Efficient Deep Reinforcement Learning Requires Regulating Overfitting","date":"2023-04-20","arxiv_id":"2304.10466","repositories_listed":0,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/efficient-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.10466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10466"}},"official":null}},{"url":"/paper/h-tsp-hierarchically-solving-the-large-scale","slug":"h-tsp-hierarchically-solving-the-large-scale","title":"H-TSP: Hierarchically Solving the Large-Scale Travelling Salesman Problem","date":"2023-04-19","arxiv_id":"2304.09395","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/h-tsp-hierarchically-solving-the-large-scale#ran","syntology_url":"https://syntology.ai/paper/2304.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09395"}},"official":{"repos":["Learning4Optimization-HUST/H-TSP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-from-passive-data-via","slug":"reinforcement-learning-from-passive-data-via","title":"Reinforcement Learning from Passive Data via Latent Intentions","date":"2023-04-10","arxiv_id":"2304.04782","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reinforcement-learning-from-passive-data-via#ran","syntology_url":"https://syntology.ai/paper/2304.04782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.04782"}},"official":{"repos":["dibyaghosh/icvf_release"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/swarm-reinforcement-learning-for-adaptive-1","slug":"swarm-reinforcement-learning-for-adaptive-1","title":"Swarm Reinforcement Learning For Adaptive Mesh Refinement","date":"2023-04-03","arxiv_id":"2304.00818","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swarm-reinforcement-learning-for-adaptive-1#ran","syntology_url":"https://syntology.ai/paper/2304.00818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.00818"}},"official":{"repos":["niklasfreymuth/asmr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-goal-reaching-reinforcement-learning","slug":"optimal-goal-reaching-reinforcement-learning","title":"Optimal Goal-Reaching Reinforcement Learning via Quasimetric Learning","date":"2023-04-03","arxiv_id":"2304.01203","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimal-goal-reaching-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2304.01203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01203"}},"official":{"repos":["quasimetric-learning/quasimetric-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mahalo-unifying-offline-reinforcement","slug":"mahalo-unifying-offline-reinforcement","title":"MAHALO: Unifying Offline Reinforcement Learning and Imitation Learning from Observations","date":"2023-03-30","arxiv_id":"2303.17156","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mahalo-unifying-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.17156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17156"}},"official":{"repos":["anqili/mahalo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reflexion-language-agents-with-verbal","slug":"reflexion-language-agents-with-verbal","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","date":"2023-03-20","arxiv_id":"2303.11366","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reflexion-language-agents-with-verbal#ran","syntology_url":"https://syntology.ai/paper/2303.11366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11366"}},"official":{"repos":["noahshinn024/reflexion"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/clip4mc-an-rl-friendly-vision-language-model","slug":"clip4mc-an-rl-friendly-vision-language-model","title":"Reinforcement Learning Friendly Vision-Language Model for Minecraft","date":"2023-03-19","arxiv_id":"2303.10571","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip4mc-an-rl-friendly-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2303.10571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10571"}},"official":{"repos":["PKU-RL/CLIP4MC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-update-to-data-ratio-minimizing-world","slug":"dynamic-update-to-data-ratio-minimizing-world","title":"Dynamic Update-to-Data Ratio: Minimizing World Model Overfitting","date":"2023-03-17","arxiv_id":"2303.10144","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-update-to-data-ratio-minimizing-world#ran","syntology_url":"https://syntology.ai/paper/2303.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10144"}},"official":{"repos":["nicolinho/dutd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kernel-density-bayesian-inverse-reinforcement","slug":"kernel-density-bayesian-inverse-reinforcement","title":"Kernel Density Bayesian Inverse Reinforcement Learning","date":"2023-03-13","arxiv_id":"2303.06827","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kernel-density-bayesian-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2303.06827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06827"}},"official":{"repos":["bee-hive/kdbirl_public"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-world-models-are-happy-with","slug":"transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","repositories_listed":1,"syntology":{"n":25,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformer-based-world-models-are-happy-with#ran","syntology_url":"https://syntology.ai/paper/2303.07109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07109"}},"official":{"repos":["jrobine/twm"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/zeroth-order-optimization-meets-human","slug":"zeroth-order-optimization-meets-human","title":"Zeroth-Order Optimization Meets Human Feedback: Provable Learning via Ranking Oracles","date":"2023-03-07","arxiv_id":"2303.03751","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/zeroth-order-optimization-meets-human#ran","syntology_url":"https://syntology.ai/paper/2303.03751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03751"}},"official":{"repos":["TZW1998/Taming-Stable-Diffusion-with-Human-Ranking-Feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-state-space-models-for-in-context-1","slug":"structured-state-space-models-for-in-context-1","title":"Structured State Space Models for In-Context Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03982","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-state-space-models-for-in-context-1#ran","syntology_url":"https://syntology.ai/paper/2303.03982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03982"}},"official":{"repos":["luchris429/popjaxrl","luchris429/s5rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-via-probabilistic","slug":"safe-reinforcement-learning-via-probabilistic","title":"Safe Reinforcement Learning via Probabilistic Logic Shields","date":"2023-03-06","arxiv_id":"2303.03226","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/safe-reinforcement-learning-via-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2303.03226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03226"}},"official":null}},{"url":"/paper/learning-to-control-autonomous-fleets-from","slug":"learning-to-control-autonomous-fleets-from","title":"Learning to Control Autonomous Fleets from Observation via Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14833","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-autonomous-fleets-from#ran","syntology_url":"https://syntology.ai/paper/2302.14833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14833"}},"official":{"repos":["carolinssc/offline-rl-amod","carolinssc/offline-rl-for-amod"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-dormant-neuron-phenomenon-in-deep","slug":"the-dormant-neuron-phenomenon-in-deep","title":"The Dormant Neuron Phenomenon in Deep Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12902","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-dormant-neuron-phenomenon-in-deep#ran","syntology_url":"https://syntology.ai/paper/2302.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12902"}},"official":{"repos":["google/dopamine"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/imitation-from-arbitrary-experience-a-dual","slug":"imitation-from-arbitrary-experience-a-dual","title":"Dual RL: Unification and New Methods for Reinforcement and Imitation Learning","date":"2023-02-16","arxiv_id":"2302.08560","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-from-arbitrary-experience-a-dual#ran","syntology_url":"https://syntology.ai/paper/2302.08560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.08560"}},"official":{"repos":["hari-sikchi/DVL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through-2","slug":"robust-deep-reinforcement-learning-through-2","title":"Regret-Based Defense in Adversarial Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06912","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-2#ran","syntology_url":"https://syntology.ai/paper/2302.06912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06912"}},"official":{"repos":["romanbelaire/robust-ccer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-decision-transformer-for-offline","slug":"constrained-decision-transformer-for-offline","title":"Constrained Decision Transformer for Offline Safe Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.07351","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constrained-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2302.07351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.07351"}},"official":{"repos":["liuzuxin/osrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-noise-filtering-with-dynamic-sparse#ran","syntology_url":"https://syntology.ai/paper/2302.06548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06548"}},"official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-penalty-based-bilevel-gradient-descent","slug":"on-penalty-based-bilevel-gradient-descent","title":"On Penalty-based Bilevel Gradient Descent Method","date":"2023-02-10","arxiv_id":"2302.05185","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-penalty-based-bilevel-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/2302.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.05185"}},"official":{"repos":["hanshen95/penalized-bilevel-gradient-descent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"0bebd871338271d895527d4ede9ee99356439ba519049d30b2df016f4e19e66c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}