{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/ran/6","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":6,"pages_in_order":12,"rows_per_page":100,"rows":[501,600],"of":1165,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2/papers/ran/1","prev":"/task/reinforcement-learning-2/papers/ran/5","next":"/task/reinforcement-learning-2/papers/ran/7","papers":[{"url":"/paper/diffusion-policies-as-an-expressive-policy","slug":"diffusion-policies-as-an-expressive-policy","title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","date":"2022-08-12","arxiv_id":"2208.06193","repositories_listed":3,"syntology":{"n":18,"n_ran":11,"n_constructed":5,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/diffusion-policies-as-an-expressive-policy#ran","syntology_url":"https://syntology.ai/paper/2208.06193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.06193"}},"official":{"repos":["zhendong-wang/diffusion-policies-for-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/robust-reinforcement-learning-using-offline","slug":"robust-reinforcement-learning-using-offline","title":"Robust Reinforcement Learning using Offline Data","date":"2022-08-10","arxiv_id":"2208.05129","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-reinforcement-learning-using-offline#ran","syntology_url":"https://syntology.ai/paper/2208.05129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.05129"}},"official":{"repos":["zaiyan-x/RFQI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-kickstarting-for-skill-transfer","slug":"hierarchical-kickstarting-for-skill-transfer","title":"Hierarchical Kickstarting for Skill Transfer in Reinforcement Learning","date":"2022-07-23","arxiv_id":"2207.11584","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-kickstarting-for-skill-transfer#ran","syntology_url":"https://syntology.ai/paper/2207.11584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.11584"}},"official":{"repos":["ucl-dark/skillhack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/log-barriers-for-safe-black-box-optimization","slug":"log-barriers-for-safe-black-box-optimization","title":"Log Barriers for Safe Black-box Optimization with Application to Safe Reinforcement Learning","date":"2022-07-21","arxiv_id":"2207.10415","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/log-barriers-for-safe-black-box-optimization#ran","syntology_url":"https://syntology.ai/paper/2207.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10415"}},"official":{"repos":["ilnura/lb_sgd","lasgroup/lbsgd-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-goal-conditioned-reinforcement","slug":"generalizing-goal-conditioned-reinforcement","title":"Generalizing Goal-Conditioned Reinforcement Learning with Variational Causal Reasoning","date":"2022-07-19","arxiv_id":"2207.09081","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizing-goal-conditioned-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.09081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09081"}},"official":{"repos":["gilgameshd/grader"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/active-exploration-for-inverse-reinforcement","slug":"active-exploration-for-inverse-reinforcement","title":"Active Exploration for Inverse Reinforcement Learning","date":"2022-07-18","arxiv_id":"2207.08645","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":3,"n_ran_checked":3,"n_instrument":7,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/active-exploration-for-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.08645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08645"}},"official":{"repos":["lasgroup/aceirl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-disentanglement-of-representations","slug":"temporal-disentanglement-of-representations","title":"Temporal Disentanglement of Representations for Improved Generalisation in Reinforcement Learning","date":"2022-07-12","arxiv_id":"2207.05480","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporal-disentanglement-of-representations#ran","syntology_url":"https://syntology.ai/paper/2207.05480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05480"}},"official":{"repos":["uoe-agents/ted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dgpo-discovering-multiple-strategies-with","slug":"dgpo-discovering-multiple-strategies-with","title":"DGPO: Discovering Multiple Strategies with Diversity-Guided Policy Optimization","date":"2022-07-12","arxiv_id":"2207.05631","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dgpo-discovering-multiple-strategies-with#ran","syntology_url":"https://syntology.ai/paper/2207.05631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05631"}},"official":{"repos":["OpenRL-Lab/DGPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/coderl-mastering-code-generation-through","slug":"coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.01780","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderl-mastering-code-generation-through#ran","syntology_url":"https://syntology.ai/paper/2207.01780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01780"}},"official":{"repos":["salesforce/coderl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stabilizing-off-policy-deep-reinforcement","slug":"stabilizing-off-policy-deep-reinforcement","title":"Stabilizing Off-Policy Deep Reinforcement Learning from Pixels","date":"2022-07-03","arxiv_id":"2207.00986","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/stabilizing-off-policy-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00986"}},"official":{"repos":["aladoro/stabilizing-off-policy-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/usher-unbiased-sampling-for-hindsight","slug":"usher-unbiased-sampling-for-hindsight","title":"USHER: Unbiased Sampling for Hindsight Experience Replay","date":"2022-07-03","arxiv_id":"2207.01115","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/usher-unbiased-sampling-for-hindsight#ran","syntology_url":"https://syntology.ai/paper/2207.01115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01115"}},"official":null}},{"url":"/paper/modular-lifelong-reinforcement-learning-via-1","slug":"modular-lifelong-reinforcement-learning-via-1","title":"Modular Lifelong Reinforcement Learning via Neural Composition","date":"2022-07-01","arxiv_id":"2207.00429","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-lifelong-reinforcement-learning-via-1#ran","syntology_url":"https://syntology.ai/paper/2207.00429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00429"}},"official":{"repos":["lifelong-ml/mendez2022modularlifelongrl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distinguishing-learning-rules-with-brain","slug":"distinguishing-learning-rules-with-brain","title":"Distinguishing Learning Rules with Brain Machine Interfaces","date":"2022-06-27","arxiv_id":"2206.13448","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distinguishing-learning-rules-with-brain#ran","syntology_url":"https://syntology.ai/paper/2206.13448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13448"}},"official":{"repos":["jacobfulano/learning-rules-with-bmi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-pretraining-vpt-learning-to-act-by","slug":"video-pretraining-vpt-learning-to-act-by","title":"Video PreTraining (VPT): Learning to Act by Watching Unlabeled Online Videos","date":"2022-06-23","arxiv_id":"2206.11795","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-pretraining-vpt-learning-to-act-by#ran","syntology_url":"https://syntology.ai/paper/2206.11795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11795"}},"official":{"repos":["openai/Video-Pre-Training"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pac-assisted-value-factorisation-with","slug":"pac-assisted-value-factorisation-with","title":"PAC: Assisted Value Factorisation with Counterfactual Predictions in Multi-Agent Reinforcement Learning","date":"2022-06-22","arxiv_id":"2206.11420","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/pac-assisted-value-factorisation-with#ran","syntology_url":"https://syntology.ai/paper/2206.11420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11420"}},"official":{"repos":["hanhananderson/pac-marl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through-1","slug":"robust-deep-reinforcement-learning-through-1","title":"Robust Deep Reinforcement Learning through Bootstrapped Opportunistic Curriculum","date":"2022-06-21","arxiv_id":"2206.10057","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-1#ran","syntology_url":"https://syntology.ai/paper/2206.10057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10057"}},"official":{"repos":["jlwu002/bcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envpool-a-highly-parallel-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.10558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10558"}},"official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/benchmarking-constraint-inference-in-inverse","slug":"benchmarking-constraint-inference-in-inverse","title":"Benchmarking Constraint Inference in Inverse Reinforcement Learning","date":"2022-06-20","arxiv_id":"2206.09670","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-constraint-inference-in-inverse#ran","syntology_url":"https://syntology.ai/paper/2206.09670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09670"}},"official":{"repos":["guiliang/cirl-benchmarks-public","guiliang/icrl-benchmarks-public"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstrapped-transformer-for-offline","slug":"bootstrapped-transformer-for-offline","title":"Bootstrapped Transformer for Offline Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08569","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bootstrapped-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.08569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08569"}},"official":null}},{"url":"/paper/smpl-simulated-industrial-manufacturing-and","slug":"smpl-simulated-industrial-manufacturing-and","title":"SMPL: Simulated Industrial Manufacturing and Process Control Learning Environments","date":"2022-06-17","arxiv_id":"2206.08851","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smpl-simulated-industrial-manufacturing-and#ran","syntology_url":"https://syntology.ai/paper/2206.08851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08851"}},"official":{"repos":["smpl-env/smpl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-population-based-reinforcement-learning","slug":"fast-population-based-reinforcement-learning","title":"Fast Population-Based Reinforcement Learning on a Single Machine","date":"2022-06-17","arxiv_id":"2206.08888","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-population-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.08888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08888"}},"official":null}},{"url":"/paper/training-discrete-deep-generative-models-via","slug":"training-discrete-deep-generative-models-via","title":"Training Discrete Deep Generative Models via Gapped Straight-Through Estimator","date":"2022-06-15","arxiv_id":"2206.07235","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-discrete-deep-generative-models-via#ran","syntology_url":"https://syntology.ai/paper/2206.07235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07235"}},"official":{"repos":["chijames/gst"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-meta-reinforcement-learners-1","slug":"transformers-are-meta-reinforcement-learners-1","title":"Transformers are Meta-Reinforcement Learners","date":"2022-06-14","arxiv_id":"2206.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transformers-are-meta-reinforcement-learners-1#ran","syntology_url":"https://syntology.ai/paper/2206.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06614"}},"official":{"repos":["luckeciano/transformers-metarl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/regularizing-a-model-based-policy-stationary","slug":"regularizing-a-model-based-policy-stationary","title":"Regularizing a Model-based Policy Stationary Distribution to Stabilize Offline Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.07166","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularizing-a-model-based-policy-stationary#ran","syntology_url":"https://syntology.ai/paper/2206.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07166"}},"official":{"repos":["shentao-yang/sdm-gan_icml2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/defending-observation-attacks-in-deep","slug":"defending-observation-attacks-in-deep","title":"Defending Observation Attacks in Deep Reinforcement Learning via Detection and Denoising","date":"2022-06-14","arxiv_id":"2206.07188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/defending-observation-attacks-in-deep#ran","syntology_url":"https://syntology.ai/paper/2206.07188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07188"}},"official":{"repos":["ZikangXiong/rl-detect-and-denoise-defense"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-approach-to-reinforcement-learning","slug":"a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-approach-to-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05825"}},"official":{"repos":["deepmind/open_spiel"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/does-self-supervised-learning-really-improve","slug":"does-self-supervised-learning-really-improve","title":"Does Self-supervised Learning Really Improve Reinforcement Learning from Pixels?","date":"2022-06-10","arxiv_id":"2206.05266","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/does-self-supervised-learning-really-improve#ran","syntology_url":"https://syntology.ai/paper/2206.05266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05266"}},"official":{"repos":["LostXine/elo-sac","lostxine/elo-rainbow"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anchor-changing-regularized-natural-policy","slug":"anchor-changing-regularized-natural-policy","title":"Anchor-Changing Regularized Natural Policy Gradient for Multi-Objective Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05357","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anchor-changing-regularized-natural-policy#ran","syntology_url":"https://syntology.ai/paper/2206.05357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05357"}},"official":{"repos":["tliu1997/arnpg-morl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mildly-conservative-q-learning-for-offline","slug":"mildly-conservative-q-learning-for-offline","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04745","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":6,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":11,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mildly-conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.04745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04745"}},"official":{"repos":["dmksjfl/mcq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/how-far-i-ll-go-offline-goal-conditioned","slug":"how-far-i-ll-go-offline-goal-conditioned","title":"How Far I'll Go: Offline Goal-Conditioned Reinforcement Learning via $f$-Advantage Regression","date":"2022-06-07","arxiv_id":"2206.03023","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-far-i-ll-go-offline-goal-conditioned#ran","syntology_url":"https://syntology.ai/paper/2206.03023","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03023"}},"official":{"repos":["jasonma2016/gofar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rorl-robust-offline-reinforcement-learning","slug":"rorl-robust-offline-reinforcement-learning","title":"RORL: Robust Offline Reinforcement Learning via Conservative Smoothing","date":"2022-06-06","arxiv_id":"2206.02829","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rorl-robust-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.02829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02829"}},"official":{"repos":["yangrui2015/rorl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-tabula-rasa-reincarnating","slug":"beyond-tabula-rasa-reincarnating","title":"Reincarnating Reinforcement Learning: Reusing Prior Computation to Accelerate Progress","date":"2022-06-03","arxiv_id":"2206.01626","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-tabula-rasa-reincarnating#ran","syntology_url":"https://syntology.ai/paper/2206.01626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01626"}},"official":{"repos":["google-research/reincarnating_rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-transformer-q-networks-for-partially","slug":"deep-transformer-q-networks-for-partially","title":"Deep Transformer Q-Networks for Partially Observable Reinforcement Learning","date":"2022-06-02","arxiv_id":"2206.01078","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-transformer-q-networks-for-partially#ran","syntology_url":"https://syntology.ai/paper/2206.01078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01078"}},"official":{"repos":["kevslinger/dtqn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-does-return-conditioned-supervised","slug":"when-does-return-conditioned-supervised","title":"When does return-conditioned supervised learning work for offline reinforcement learning?","date":"2022-06-02","arxiv_id":"2206.01079","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/when-does-return-conditioned-supervised#ran","syntology_url":"https://syntology.ai/paper/2206.01079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01079"}},"official":{"repos":["davidbrandfonbrener/rcsl-paper"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-reward-poisoning-attacks-on-online#ran","syntology_url":"https://syntology.ai/paper/2205.14842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14842"}},"official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-a-terminator","slug":"reinforcement-learning-with-a-terminator","title":"Reinforcement Learning with a Terminator","date":"2022-05-30","arxiv_id":"2205.15376","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-a-terminator#ran","syntology_url":"https://syntology.ai/paper/2205.15376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.15376"}},"official":{"repos":["guytenn/terminator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dep-rl-embodied-exploration-for-reinforcement","slug":"dep-rl-embodied-exploration-for-reinforcement","title":"DEP-RL: Embodied Exploration for Reinforcement Learning in Overactuated and Musculoskeletal Systems","date":"2022-05-30","arxiv_id":"2206.00484","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dep-rl-embodied-exploration-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.00484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00484"}},"official":{"repos":["martius-lab/depRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-robustness-of-safe-reinforcement","slug":"on-the-robustness-of-safe-reinforcement","title":"On the Robustness of Safe Reinforcement Learning under Observational Perturbations","date":"2022-05-29","arxiv_id":"2205.14691","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-the-robustness-of-safe-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14691"}},"official":{"repos":["liuzuxin/safe-rl-robustness"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-continual-reinforcement","slug":"task-agnostic-continual-reinforcement","title":"Task-Agnostic Continual Reinforcement Learning: Gaining Insights and Overcoming Challenges","date":"2022-05-28","arxiv_id":"2205.14495","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/task-agnostic-continual-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.14495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14495"}},"official":{"repos":["amazon-science/replay-based-recurrent-rl","amazon-research/replay-based-recurrent-rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fedformer-contextual-federation-with","slug":"fedformer-contextual-federation-with","title":"FedFormer: Contextual Federation with Attention in Reinforcement Learning","date":"2022-05-27","arxiv_id":"2205.13697","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedformer-contextual-federation-with#ran","syntology_url":"https://syntology.ai/paper/2205.13697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13697"}},"official":{"repos":["liamhebert/FedFormer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tiered-reinforcement-learning-pessimism-in","slug":"tiered-reinforcement-learning-pessimism-in","title":"Tiered Reinforcement Learning: Pessimism in the Face of Uncertainty and Constant Regret","date":"2022-05-25","arxiv_id":"2205.12418","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/tiered-reinforcement-learning-pessimism-in#ran","syntology_url":"https://syntology.ai/paper/2205.12418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12418"}},"official":{"repos":["jiaweihhuang/tiered-rl-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/skill-machines-temporal-logic-composition-in","slug":"skill-machines-temporal-logic-composition-in","title":"Skill Machines: Temporal Logic Skill Composition in Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.12532","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/skill-machines-temporal-logic-composition-in#ran","syntology_url":"https://syntology.ai/paper/2205.12532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12532"}},"official":{"repos":["geraudnt/skill_machines"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/history-compression-via-language-models-in","slug":"history-compression-via-language-models-in","title":"History Compression via Language Models in Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12258","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/history-compression-via-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2205.12258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12258"}},"official":{"repos":["ml-jku/helm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-uncertainty-for-exploration-in-1","slug":"reward-uncertainty-for-exploration-in-1","title":"Reward Uncertainty for Exploration in Preference-based Reinforcement Learning","date":"2022-05-24","arxiv_id":"2205.12401","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-uncertainty-for-exploration-in-1#ran","syntology_url":"https://syntology.ai/paper/2205.12401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12401"}},"official":null}},{"url":"/paper/an-evaluation-study-of-intrinsic-motivation","slug":"an-evaluation-study-of-intrinsic-motivation","title":"An Evaluation Study of Intrinsic Motivation Techniques applied to Reinforcement Learning over Hard Exploration Environments","date":"2022-05-23","arxiv_id":"2205.11184","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-evaluation-study-of-intrinsic-motivation#ran","syntology_url":"https://syntology.ai/paper/2205.11184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11184"}},"official":{"repos":["aklein1995/intrinsic_motivation_techniques_study"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/memory-efficient-reinforcement-learning-with","slug":"memory-efficient-reinforcement-learning-with","title":"Memory-efficient Reinforcement Learning with Value-based Knowledge Consolidation","date":"2022-05-22","arxiv_id":"2205.10868","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memory-efficient-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2205.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10868"}},"official":{"repos":["qlan3/MeDQN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reachability-constrained-reinforcement","slug":"reachability-constrained-reinforcement","title":"Reachability Constrained Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07536","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reachability-constrained-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07536"}},"official":{"repos":["mahaitongdae/Reachability_Constrained_RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cliff-diving-exploring-reward-surfaces-in","slug":"cliff-diving-exploring-reward-surfaces-in","title":"Cliff Diving: Exploring Reward Surfaces in Reinforcement Learning Environments","date":"2022-05-14","arxiv_id":"2205.07015","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cliff-diving-exploring-reward-surfaces-in#ran","syntology_url":"https://syntology.ai/paper/2205.07015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.07015"}},"official":null}},{"url":"/paper/a-state-distribution-matching-approach-to-non","slug":"a-state-distribution-matching-approach-to-non","title":"A State-Distribution Matching Approach to Non-Episodic Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05212","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-state-distribution-matching-approach-to-non#ran","syntology_url":"https://syntology.ai/paper/2205.05212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05212"}},"official":{"repos":["architsharma97/medal"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-risk-averse-reinforcement-learning","slug":"efficient-risk-averse-reinforcement-learning","title":"Efficient Risk-Averse Reinforcement Learning","date":"2022-05-10","arxiv_id":"2205.05138","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-risk-averse-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2205.05138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05138"}},"official":{"repos":["ido90/CeSoR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cclf-a-contrastive-curiosity-driven-learning","slug":"cclf-a-contrastive-curiosity-driven-learning","title":"CCLF: A Contrastive-Curiosity-Driven Learning Framework for Sample-Efficient Reinforcement Learning","date":"2022-05-02","arxiv_id":"2205.00943","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cclf-a-contrastive-curiosity-driven-learning#ran","syntology_url":"https://syntology.ai/paper/2205.00943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00943"}},"official":{"repos":["csun001/cclf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ttopt-a-maximum-volume-quantized-tensor-train","slug":"ttopt-a-maximum-volume-quantized-tensor-train","title":"TTOpt: A Maximum Volume Quantized Tensor Train-based Optimization and its Application to Reinforcement Learning","date":"2022-04-30","arxiv_id":"2205.00293","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":6,"n_instrument":8,"n_unverified":3,"n_honours":6,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 6 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ttopt-a-maximum-volume-quantized-tensor-train#ran","syntology_url":"https://syntology.ai/paper/2205.00293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00293"}},"official":{"repos":["andreichertkov/ttopt"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/markov-abstractions-for-pac-reinforcement","slug":"markov-abstractions-for-pac-reinforcement","title":"Markov Abstractions for PAC Reinforcement Learning in Non-Markov Decision Processes","date":"2022-04-29","arxiv_id":"2205.01053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/markov-abstractions-for-pac-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2205.01053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01053"}},"official":{"repos":["whitemech/markov-abstractions-code-ijcai22"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rambo-rl-robust-adversarial-model-based","slug":"rambo-rl-robust-adversarial-model-based","title":"RAMBO-RL: Robust Adversarial Model-Based Offline Reinforcement Learning","date":"2022-04-26","arxiv_id":"2204.12581","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rambo-rl-robust-adversarial-model-based#ran","syntology_url":"https://syntology.ai/paper/2204.12581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.12581"}},"official":{"repos":["marc-rigter/rambo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hypernca-growing-developmental-networks-with","slug":"hypernca-growing-developmental-networks-with","title":"HyperNCA: Growing Developmental Networks with Neural Cellular Automata","date":"2022-04-25","arxiv_id":"2204.11674","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypernca-growing-developmental-networks-with#ran","syntology_url":"https://syntology.ai/paper/2204.11674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11674"}},"official":null}},{"url":"/paper/multi-objective-pointer-network-for","slug":"multi-objective-pointer-network-for","title":"Multi-objective Pointer Network for Combinatorial Optimization","date":"2022-04-25","arxiv_id":"2204.11860","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-objective-pointer-network-for#ran","syntology_url":"https://syntology.ai/paper/2204.11860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11860"}},"official":{"repos":["gaoly/mopn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coptidice-offline-constrained-reinforcement-1","slug":"coptidice-offline-constrained-reinforcement-1","title":"COptiDICE: Offline Constrained Reinforcement Learning via Stationary Distribution Correction Estimation","date":"2022-04-19","arxiv_id":"2204.08957","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coptidice-offline-constrained-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2204.08957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08957"}},"official":{"repos":["deepmind/constrained_optidice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue","slug":"chai-a-chatbot-ai-for-task-oriented-dialogue","title":"CHAI: A CHatbot AI for Task-Oriented Dialogue with Offline Reinforcement Learning","date":"2022-04-18","arxiv_id":"2204.08426","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue#ran","syntology_url":"https://syntology.ai/paper/2204.08426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.08426"}},"official":{"repos":["siddharthverma314/chai-naacl-2022"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-question-rewriting-help-conversational","slug":"can-question-rewriting-help-conversational","title":"Can Question Rewriting Help Conversational Question Answering?","date":"2022-04-13","arxiv_id":"2204.06239","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-question-rewriting-help-conversational#ran","syntology_url":"https://syntology.ai/paper/2204.06239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06239"}},"official":{"repos":["hltchkust/cqr4cqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/training-a-helpful-and-harmless-assistant","slug":"training-a-helpful-and-harmless-assistant","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","date":"2022-04-12","arxiv_id":"2204.05862","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-a-helpful-and-harmless-assistant#ran","syntology_url":"https://syntology.ai/paper/2204.05862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05862"}},"official":{"repos":["anthropics/hh-rlhf"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/grounding-hindsight-instructions-in-multi","slug":"grounding-hindsight-instructions-in-multi","title":"Grounding Hindsight Instructions in Multi-Goal Reinforcement Learning for Robotics","date":"2022-04-08","arxiv_id":"2204.04308","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-hindsight-instructions-in-multi#ran","syntology_url":"https://syntology.ai/paper/2204.04308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04308"}},"official":{"repos":["knowledgetechnologyuhh/hipss"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/asynchronous-reinforcement-learning-for-real","slug":"asynchronous-reinforcement-learning-for-real","title":"Asynchronous Reinforcement Learning for Real-Time Control of Physical Robots","date":"2022-03-23","arxiv_id":"2203.12759","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asynchronous-reinforcement-learning-for-real#ran","syntology_url":"https://syntology.ai/paper/2203.12759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.12759"}},"official":{"repos":["yufengyuan/ur5_async_rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/teachable-reinforcement-learning-via-advice-1","slug":"teachable-reinforcement-learning-via-advice-1","title":"Teachable Reinforcement Learning via Advice Distillation","date":"2022-03-19","arxiv_id":"2203.11197","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teachable-reinforcement-learning-via-advice-1#ran","syntology_url":"https://syntology.ai/paper/2203.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11197"}},"official":{"repos":["rll-research/teachable"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pmic-improving-multi-agent-reinforcement-1","slug":"pmic-improving-multi-agent-reinforcement-1","title":"PMIC: Improving Multi-Agent Reinforcement Learning with Progressive Mutual Information Collaboration","date":"2022-03-16","arxiv_id":"2203.08553","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmic-improving-multi-agent-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2203.08553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08553"}},"official":{"repos":["yeshenpy/pmic"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/latent-variable-advantage-weighted-policy","slug":"latent-variable-advantage-weighted-policy","title":"Latent-Variable Advantage-Weighted Policy Optimization for Offline RL","date":"2022-03-16","arxiv_id":"2203.08949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-variable-advantage-weighted-policy#ran","syntology_url":"https://syntology.ai/paper/2203.08949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08949"}},"official":null}},{"url":"/paper/zipfian-environments-for-reinforcement","slug":"zipfian-environments-for-reinforcement","title":"Zipfian environments for Reinforcement Learning","date":"2022-03-15","arxiv_id":"2203.08222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zipfian-environments-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2203.08222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08222"}},"official":{"repos":["deepmind/zipfian_environments"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-entity-1","slug":"deep-reinforcement-learning-for-entity-1","title":"Deep Reinforcement Learning for Entity Alignment","date":"2022-03-07","arxiv_id":"2203.03315","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-entity-1#ran","syntology_url":"https://syntology.ai/paper/2203.03315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03315"}},"official":{"repos":["guolingbing/rlea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/influencing-long-term-behavior-in-multiagent","slug":"influencing-long-term-behavior-in-multiagent","title":"Influencing Long-Term Behavior in Multiagent Reinforcement Learning","date":"2022-03-07","arxiv_id":"2203.03535","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":3,"n_ran_checked":4,"n_instrument":6,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/influencing-long-term-behavior-in-multiagent#ran","syntology_url":"https://syntology.ai/paper/2203.03535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03535"}},"official":{"repos":["dkkim93/further"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-offline-reinforcement-learning","slug":"a-survey-on-offline-reinforcement-learning","title":"A Survey on Offline Reinforcement Learning: Taxonomy, Review, and Open Problems","date":"2022-03-02","arxiv_id":"2203.01387","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-survey-on-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2203.01387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01387"}},"official":{"repos":["larocs/offline-rl-suvey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantum-deep-reinforcement-learning-for-robot","slug":"quantum-deep-reinforcement-learning-for-robot","title":"Quantum Deep Reinforcement Learning for Robot Navigation Tasks","date":"2022-02-24","arxiv_id":"2202.12180","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/quantum-deep-reinforcement-learning-for-robot#ran","syntology_url":"https://syntology.ai/paper/2202.12180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12180"}},"official":{"repos":["dfki-ric-quantum/qdrl-turtlebot-env","dfki-ric-quantum/qdrl-turtlebot-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-transferable-reward-for-query-object-1","slug":"learning-transferable-reward-for-query-object-1","title":"Learning Transferable Reward for Query Object Localization with Policy Adaptation","date":"2022-02-24","arxiv_id":"2202.12403","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-transferable-reward-for-query-object-1#ran","syntology_url":"https://syntology.ai/paper/2202.12403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12403"}},"official":{"repos":["litingfeng/localization-by-ordembed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transdreamer-reinforcement-learning-with-1","slug":"transdreamer-reinforcement-learning-with-1","title":"TransDreamer: Reinforcement Learning with Transformer World Models","date":"2022-02-19","arxiv_id":"2202.09481","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transdreamer-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2202.09481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.09481"}},"official":null}},{"url":"/paper/cadre-a-cascade-deep-reinforcement-learning","slug":"cadre-a-cascade-deep-reinforcement-learning","title":"CADRE: A Cascade Deep Reinforcement Learning Framework for Vision-based Autonomous Urban Driving","date":"2022-02-17","arxiv_id":"2202.08557","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":7,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cadre-a-cascade-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2202.08557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.08557"}},"official":{"repos":["BIT-MCS/Cadre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vrl3-a-data-driven-framework-for-visual-deep","slug":"vrl3-a-data-driven-framework-for-visual-deep","title":"VRL3: A Data-Driven Framework for Visual Deep Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.10324","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrl3-a-data-driven-framework-for-visual-deep#ran","syntology_url":"https://syntology.ai/paper/2202.10324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10324"}},"official":{"repos":["facebookresearch/drqv2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-by-imagining-the-1","slug":"safe-reinforcement-learning-by-imagining-the-1","title":"Safe Reinforcement Learning by Imagining the Near Future","date":"2022-02-15","arxiv_id":"2202.07789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-reinforcement-learning-by-imagining-the-1#ran","syntology_url":"https://syntology.ai/paper/2202.07789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.07789"}},"official":{"repos":["gwthomas/safe-mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-shapley-value-in-machine-learning","slug":"the-shapley-value-in-machine-learning","title":"The Shapley Value in Machine Learning","date":"2022-02-11","arxiv_id":"2202.05594","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-shapley-value-in-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2202.05594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.05594"}},"official":{"repos":["benedekrozemberczki/shapley"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextualize-me-the-case-for-context-in","slug":"contextualize-me-the-case-for-context-in","title":"Contextualize Me -- The Case for Context in Reinforcement Learning","date":"2022-02-09","arxiv_id":"2202.04500","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextualize-me-the-case-for-context-in#ran","syntology_url":"https://syntology.ai/paper/2202.04500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04500"}},"official":{"repos":["automl/CARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-sparse-rewards-1","slug":"reinforcement-learning-with-sparse-rewards-1","title":"Reinforcement Learning with Sparse Rewards using Guidance from Offline Demonstration","date":"2022-02-09","arxiv_id":"2202.04628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-sparse-rewards-1#ran","syntology_url":"https://syntology.ai/paper/2202.04628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04628"}},"official":{"repos":["desikrengarajan/logo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-nonparametrics-for-offline-skill","slug":"bayesian-nonparametrics-for-offline-skill","title":"Bayesian Nonparametrics for Offline Skill Discovery","date":"2022-02-09","arxiv_id":"2202.04675","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bayesian-nonparametrics-for-offline-skill#ran","syntology_url":"https://syntology.ai/paper/2202.04675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04675"}},"official":{"repos":["layer6ai-labs/bnpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/approximating-gradients-for-differentiable","slug":"approximating-gradients-for-differentiable","title":"Approximating Gradients for Differentiable Quality Diversity in Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03666","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/approximating-gradients-for-differentiable#ran","syntology_url":"https://syntology.ai/paper/2202.03666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03666"}},"official":{"repos":["icaros-usc/dqd-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-synthetic-environments-and-reward-1","slug":"learning-synthetic-environments-and-reward-1","title":"Learning Synthetic Environments and Reward Networks for Reinforcement Learning","date":"2022-02-06","arxiv_id":"2202.02790","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-synthetic-environments-and-reward-1#ran","syntology_url":"https://syntology.ai/paper/2202.02790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02790"}},"official":{"repos":["automl/learning_environments"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarially-trained-actor-critic-for","slug":"adversarially-trained-actor-critic-for","title":"Adversarially Trained Actor Critic for Offline Reinforcement Learning","date":"2022-02-05","arxiv_id":"2202.02446","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarially-trained-actor-critic-for#ran","syntology_url":"https://syntology.ai/paper/2202.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02446"}},"official":{"repos":["microsoft/atac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cic-contrastive-intrinsic-control-for-1","slug":"cic-contrastive-intrinsic-control-for-1","title":"CIC: Contrastive Intrinsic Control for Unsupervised Skill Discovery","date":"2022-02-01","arxiv_id":"2202.00161","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cic-contrastive-intrinsic-control-for-1#ran","syntology_url":"https://syntology.ai/paper/2202.00161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.00161"}},"official":null}},{"url":"/paper/dns-determinantal-point-process-based-neural","slug":"dns-determinantal-point-process-based-neural","title":"DNS: Determinantal Point Process Based Neural Network Sampler for Ensemble Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13357","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dns-determinantal-point-process-based-neural#ran","syntology_url":"https://syntology.ai/paper/2201.13357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13357"}},"official":{"repos":["IntelLabs/DNS"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/don-t-change-the-algorithm-change-the-data","slug":"don-t-change-the-algorithm-change-the-data","title":"Don't Change the Algorithm, Change the Data: Exploratory Data for Offline Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13425","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/don-t-change-the-algorithm-change-the-data#ran","syntology_url":"https://syntology.ai/paper/2201.13425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.13425"}},"official":{"repos":["denisyarats/exorl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-reinforcement-learning-policies","slug":"explaining-reinforcement-learning-policies","title":"Explaining Reinforcement Learning Policies through Counterfactual Trajectories","date":"2022-01-29","arxiv_id":"2201.12462","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/explaining-reinforcement-learning-policies#ran","syntology_url":"https://syntology.ai/paper/2201.12462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12462"}},"official":{"repos":["juliusfrost/cfrl-rllib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-variational-policy-optimization","slug":"constrained-variational-policy-optimization","title":"Constrained Variational Policy Optimization for Safe Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.11927","repositories_listed":2,"syntology":{"n":11,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/constrained-variational-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2201.11927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11927"}},"official":{"repos":["liuzuxin/cvpo-safe-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/mask-based-latent-reconstruction-for","slug":"mask-based-latent-reconstruction-for","title":"Mask-based Latent Reconstruction for Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.12096","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":4,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mask-based-latent-reconstruction-for#ran","syntology_url":"https://syntology.ai/paper/2201.12096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12096"}},"official":{"repos":["microsoft/Mask-based-Latent-Reconstruction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/can-wikipedia-help-offline-reinforcement","slug":"can-wikipedia-help-offline-reinforcement","title":"Can Wikipedia Help Offline Reinforcement Learning?","date":"2022-01-28","arxiv_id":"2201.12122","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-wikipedia-help-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2201.12122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12122"}},"official":{"repos":["machelreid/can-wikipedia-help-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-safe-reinforcement-learning-with-a","slug":"towards-safe-reinforcement-learning-with-a","title":"Towards Safe Reinforcement Learning with a Safety Editor Policy","date":"2022-01-28","arxiv_id":"2201.12427","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-safe-reinforcement-learning-with-a#ran","syntology_url":"https://syntology.ai/paper/2201.12427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12427"}},"official":{"repos":["hnyu/seditor"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/exploration-with-a-finite-brain","slug":"exploration-with-a-finite-brain","title":"Modeling Human Exploration Through Resource-Rational Reinforcement Learning","date":"2022-01-27","arxiv_id":"2201.11817","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploration-with-a-finite-brain#ran","syntology_url":"https://syntology.ai/paper/2201.11817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11817"}},"official":{"repos":["marcelbinz/resource-rational-reinforcement-learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-policy-optimization-via-bayesian-1","slug":"constrained-policy-optimization-via-bayesian-1","title":"Constrained Policy Optimization via Bayesian World Models","date":"2022-01-24","arxiv_id":"2201.09802","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/constrained-policy-optimization-via-bayesian-1#ran","syntology_url":"https://syntology.ai/paper/2201.09802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.09802"}},"official":{"repos":["yardenas/la-mbda"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/environment-generation-for-zero-shot-1","slug":"environment-generation-for-zero-shot-1","title":"Environment Generation for Zero-Shot Compositional Reinforcement Learning","date":"2022-01-21","arxiv_id":"2201.08896","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/environment-generation-for-zero-shot-1#ran","syntology_url":"https://syntology.ai/paper/2201.08896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08896"}},"official":{"repos":["google-research/google-research"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-deep-reinforcement-learning-5","slug":"sample-efficient-deep-reinforcement-learning-5","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","date":"2022-01-05","arxiv_id":"2201.01666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-deep-reinforcement-learning-5#ran","syntology_url":"https://syntology.ai/paper/2201.01666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01666"}},"official":{"repos":["montrealrobotics/iv_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/settling-the-bias-and-variance-of-meta","slug":"settling-the-bias-and-variance-of-meta","title":"A Theoretical Understanding of Gradient Bias in Meta-Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15400","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/settling-the-bias-and-variance-of-meta#ran","syntology_url":"https://syntology.ai/paper/2112.15400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.15400"}},"official":{"repos":["Benjamin-eecs/Theoretical-GMRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/exponential-family-model-based-reinforcement","slug":"exponential-family-model-based-reinforcement","title":"Exponential Family Model-Based Reinforcement Learning via Score Matching","date":"2021-12-28","arxiv_id":"2112.14195","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exponential-family-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.14195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.14195"}},"official":{"repos":["anmolkabra/score-matching-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-dynamic-convex","slug":"reinforcement-learning-with-dynamic-convex","title":"Reinforcement Learning with Dynamic Convex Risk Measures","date":"2021-12-26","arxiv_id":"2112.13414","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-dynamic-convex#ran","syntology_url":"https://syntology.ai/paper/2112.13414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.13414"}},"official":{"repos":["acoache/rl-dynamicconvexrisk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autonomous-reinforcement-learning-formalism-1","slug":"autonomous-reinforcement-learning-formalism-1","title":"Autonomous Reinforcement Learning: Formalism and Benchmarking","date":"2021-12-17","arxiv_id":"2112.09605","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autonomous-reinforcement-learning-formalism-1#ran","syntology_url":"https://syntology.ai/paper/2112.09605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.09605"}},"official":{"repos":["architsharma97/earl_benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-reinforcement-learning-in","slug":"unsupervised-reinforcement-learning-in","title":"Unsupervised Reinforcement Learning in Multiple Environments","date":"2021-12-16","arxiv_id":"2112.08746","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unsupervised-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/2112.08746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.08746"}},"official":{"repos":["muttimirco/alphamepol"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finrl-meta-a-universe-of-near-real-market","slug":"finrl-meta-a-universe-of-near-real-market","title":"FinRL-Meta: A Universe of Near-Real Market Environments for Data-Driven Deep Reinforcement Learning in Quantitative Finance","date":"2021-12-13","arxiv_id":"2112.06753","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/finrl-meta-a-universe-of-near-real-market#ran","syntology_url":"https://syntology.ai/paper/2112.06753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06753"}},"official":{"repos":["ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"4d49afdd6ea04b592011a0b4377df24336299a11c612f34785c29937e139cb30","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}