{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/ran/7","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":7,"pages_in_order":12,"rows_per_page":100,"rows":[601,700],"of":1175,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning/papers/ran/1","prev":"/task/reinforcement-learning/papers/ran/6","next":"/task/reinforcement-learning/papers/ran/8","papers":[{"url":"/paper/memory-based-deep-reinforcement-learning-for-1","slug":"memory-based-deep-reinforcement-learning-for-1","title":"Memory-based Deep Reinforcement Learning for POMDPs","date":"2021-02-24","arxiv_id":"2102.12344","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/memory-based-deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2102.12344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.12344"}},"official":{"repos":["LinghengMeng/LSTM-TD3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/information-directed-reward-learning-for","slug":"information-directed-reward-learning-for","title":"Information Directed Reward Learning for Reinforcement Learning","date":"2021-02-24","arxiv_id":"2102.12466","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/information-directed-reward-learning-for#ran","syntology_url":"https://syntology.ai/paper/2102.12466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.12466"}},"official":{"repos":["david-lindner/idrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/program-synthesis-guided-reinforcement","slug":"program-synthesis-guided-reinforcement","title":"Program Synthesis Guided Reinforcement Learning for Partially Observed Environments","date":"2021-02-22","arxiv_id":"2102.11137","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/program-synthesis-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2102.11137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.11137"}},"official":{"repos":["yycdavid/program-synthesis-guided-rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-doubly-constrained-batch","slug":"continuous-doubly-constrained-batch","title":"Continuous Doubly Constrained Batch Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09225","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/continuous-doubly-constrained-batch#ran","syntology_url":"https://syntology.ai/paper/2102.09225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.09225"}},"official":{"repos":["amazon-research/cdc-batch-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-rational-networks#ran","syntology_url":"https://syntology.ai/paper/2102.09407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.09407"}},"official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-multi-agent-reinforcement-learning","slug":"scaling-multi-agent-reinforcement-learning","title":"Scaling Multi-Agent Reinforcement Learning with Selective Parameter Sharing","date":"2021-02-15","arxiv_id":"2102.07475","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/scaling-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.07475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07475"}},"official":{"repos":["uoe-agents/seps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/domain-adaptation-in-reinforcement-learning","slug":"domain-adaptation-in-reinforcement-learning","title":"Domain Adaptation In Reinforcement Learning Via Latent Unified State Representation","date":"2021-02-10","arxiv_id":"2102.05714","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/domain-adaptation-in-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.05714"}},"official":{"repos":["KarlXing/LUSR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameter-tricks-in-multi-agent","slug":"hyperparameter-tricks-in-multi-agent","title":"Rethinking the Implementation Matters in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-06","arxiv_id":"2102.03479","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameter-tricks-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2102.03479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03479"}},"official":{"repos":["hijkzzz/pymarl2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-agent-reinforcement-learning-with","slug":"multi-agent-reinforcement-learning-with","title":"Multi-Agent Reinforcement Learning with Temporal Logic Specifications","date":"2021-02-01","arxiv_id":"2102.00582","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.00582"}},"official":{"repos":["lrhammond/almanac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-on-state-1","slug":"robust-reinforcement-learning-on-state-1","title":"Robust Reinforcement Learning on State Observations with Learned Optimal Adversary","date":"2021-01-21","arxiv_id":"2101.08452","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/robust-reinforcement-learning-on-state-1#ran","syntology_url":"https://syntology.ai/paper/2101.08452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08452"}},"official":{"repos":["huanzhang12/ATLA_robust_RL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/solving-common-payoff-games-with-approximate","slug":"solving-common-payoff-games-with-approximate","title":"Solving Common-Payoff Games with Approximate Policy Iteration","date":"2021-01-11","arxiv_id":"2101.04237","repositories_listed":2,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/solving-common-payoff-games-with-approximate#ran","syntology_url":"https://syntology.ai/paper/2101.04237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.04237"}},"official":{"repos":["ssokota/capi","ssokota/tiny-hanabi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-latent-flow-1","slug":"reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","arxiv_id":"2101.01857","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-latent-flow-1#ran","syntology_url":"https://syntology.ai/paper/2101.01857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.01857"}},"official":{"repos":["WendyShang/flare","WendyShang/dqn_zoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/augmenting-policy-learning-with-routines","slug":"augmenting-policy-learning-with-routines","title":"Augmenting Policy Learning with Routines Discovered from a Single Demonstration","date":"2020-12-23","arxiv_id":"2012.12469","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-policy-learning-with-routines#ran","syntology_url":"https://syntology.ai/paper/2012.12469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.12469"}},"official":{"repos":["sjtuytc/-AAAI21-RoutineAugmentedPolicyLearning-RAPL-","sjtuytc/AAAI21-RoutineAugmentedPolicyLearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/high-throughput-synchronous-deep-rl-1","slug":"high-throughput-synchronous-deep-rl-1","title":"High-Throughput Synchronous Deep RL","date":"2020-12-17","arxiv_id":"2012.09849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/high-throughput-synchronous-deep-rl-1#ran","syntology_url":"https://syntology.ai/paper/2012.09849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09849"}},"official":{"repos":["IouJenLiu/HTS-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-constrained-reinforcement-learning","slug":"inverse-constrained-reinforcement-learning","title":"Inverse Constrained Reinforcement Learning","date":"2020-11-19","arxiv_id":"2011.09999","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/inverse-constrained-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2011.09999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.09999"}},"official":{"repos":["shehryar-malik/icrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-reinforcement-learning-policies-for","slug":"scalable-reinforcement-learning-policies-for","title":"Scalable Reinforcement Learning Policies for Multi-Agent Control","date":"2020-11-16","arxiv_id":"2011.08055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-reinforcement-learning-policies-for#ran","syntology_url":"https://syntology.ai/paper/2011.08055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.08055"}},"official":{"repos":["christopher-hsu/scalableMARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-to-new-actions-in-1","slug":"generalization-to-new-actions-in-1","title":"Generalization to New Actions in Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.01928","repositories_listed":2,"syntology":{"n":21,"n_ran":18,"n_constructed":14,"n_ran_checked":16,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":14,"n_pointer_only":20,"phrase":"18 ran (of which 14 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/generalization-to-new-actions-in-1#ran","syntology_url":"https://syntology.ai/paper/2011.01928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01928"}},"official":{"repos":["clvrai/new-actions-rl","clvrai/create"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":14,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/instance-based-generalization-in","slug":"instance-based-generalization-in","title":"Instance based Generalization in Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01089","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instance-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2011.01089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01089"}},"official":{"repos":["MartinBertran/InstanceAgnosticPolicyEnsembles"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-overview-of-multi-agent-reinforcement","slug":"an-overview-of-multi-agent-reinforcement","title":"Game-Theoretic Multiagent Reinforcement Learning","date":"2020-11-01","arxiv_id":"2011.00583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-overview-of-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2011.00583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00583"}},"official":null}},{"url":"/paper/pomo-policy-optimization-with-multiple-optima","slug":"pomo-policy-optimization-with-multiple-optima","title":"POMO: Policy Optimization with Multiple Optima for Reinforcement Learning","date":"2020-10-30","arxiv_id":"2010.16011","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/pomo-policy-optimization-with-multiple-optima#ran","syntology_url":"https://syntology.ai/paper/2010.16011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.16011"}},"official":{"repos":["yd-kwon/POMO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/recovery-rl-safe-reinforcement-learning-with","slug":"recovery-rl-safe-reinforcement-learning-with","title":"Recovery RL: Safe Reinforcement Learning with Learned Recovery Zones","date":"2020-10-29","arxiv_id":"2010.15920","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recovery-rl-safe-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2010.15920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.15920"}},"official":null}},{"url":"/paper/implicit-under-parameterization-inhibits-data-1","slug":"implicit-under-parameterization-inhibits-data-1","title":"Implicit Under-Parameterization Inhibits Data-Efficient Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-under-parameterization-inhibits-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.14498"}},"official":null}},{"url":"/paper/bridging-imagination-and-reality-for-model","slug":"bridging-imagination-and-reality-for-model","title":"Bridging Imagination and Reality for Model-Based Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12142","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-imagination-and-reality-for-model#ran","syntology_url":"https://syntology.ai/paper/2010.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12142"}},"official":{"repos":["Mehooz/BIRD_code"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/drift-detection-in-episodic-data-detect-when-1","slug":"drift-detection-in-episodic-data-detect-when-1","title":"Detecting Rewards Deterioration in Episodic Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11660","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drift-detection-in-episodic-data-detect-when-1#ran","syntology_url":"https://syntology.ai/paper/2010.11660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11660"}},"official":{"repos":["ido90/Rewards-Deterioration-Detection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-optimization-of","slug":"reinforcement-learning-for-optimization-of","title":"Reinforcement Learning for Optimization of COVID-19 Mitigation policies","date":"2020-10-20","arxiv_id":"2010.10560","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-for-optimization-of#ran","syntology_url":"https://syntology.ai/paper/2010.10560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10560"}},"official":{"repos":["SonyAI/PandemicSimulator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/d2rl-deep-dense-architectures-in-1","slug":"d2rl-deep-dense-architectures-in-1","title":"D2RL: Deep Dense Architectures in Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09163","repositories_listed":4,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/d2rl-deep-dense-architectures-in-1#ran","syntology_url":"https://syntology.ai/paper/2010.09163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09163"}},"official":{"repos":["pairlab/d2rl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","listed","official"]}}},{"url":"/paper/knowledge-guided-open-attribute-value","slug":"knowledge-guided-open-attribute-value","title":"Knowledge-guided Open Attribute Value Extraction with Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09189","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/knowledge-guided-open-attribute-value#ran","syntology_url":"https://syntology.ai/paper/2010.09189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09189"}},"official":{"repos":["yeliu0930/Knowledge-guided-Open-Attribute-Value-Extraction-with-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-policy-optimization-with","slug":"model-based-policy-optimization-with","title":"Model-based Policy Optimization with Unsupervised Model Adaptation","date":"2020-10-19","arxiv_id":"2010.09546","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/model-based-policy-optimization-with#ran","syntology_url":"https://syntology.ai/paper/2010.09546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09546"}},"official":{"repos":["RockySJ/ampo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameter-auto-tuning-in-self-supervised","slug":"hyperparameter-auto-tuning-in-self-supervised","title":"Hyperparameter Auto-tuning in Self-Supervised Robotic Learning","date":"2020-10-16","arxiv_id":"2010.08252","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hyperparameter-auto-tuning-in-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.08252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.08252"}},"official":{"repos":["birlrobotics/rlkit_autotune"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-contrastive-representation-learning","slug":"masked-contrastive-representation-learning","title":"Masked Contrastive Representation Learning for Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.07470","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/masked-contrastive-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2010.07470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07470"}},"official":{"repos":["teslacool/m-curl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with-1","slug":"multi-task-deep-reinforcement-learning-with-1","title":"Knowledge Transfer in Multi-Task Deep Reinforcement Learning for Continuous Control","date":"2020-10-15","arxiv_id":"2010.07494","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2010.07494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07494"}},"official":null}},{"url":"/paper/a-game-theoretic-analysis-of-networked-system","slug":"a-game-theoretic-analysis-of-networked-system","title":"A game-theoretic analysis of networked system control for common-pool resource management using multi-agent reinforcement learning","date":"2020-10-15","arxiv_id":"2010.07777","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-game-theoretic-analysis-of-networked-system#ran","syntology_url":"https://syntology.ai/paper/2010.07777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07777"}},"official":{"repos":["instadeepai/EGTA-NMARL"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/human-centric-dialog-training-via-offline","slug":"human-centric-dialog-training-via-offline","title":"Human-centric Dialog Training via Offline Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05848","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-dialog-training-via-offline#ran","syntology_url":"https://syntology.ai/paper/2010.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05848"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/contrastive-explanations-for-reinforcement-2","slug":"contrastive-explanations-for-reinforcement-2","title":"Contrastive Explanations for Reinforcement Learning via Embedded Self Predictions","date":"2020-10-11","arxiv_id":"2010.05180","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/contrastive-explanations-for-reinforcement-2#ran","syntology_url":"https://syntology.ai/paper/2010.05180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05180"}},"official":{"repos":["SuerpX/Embedded-Self-Predictions"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-random-delays-1","slug":"reinforcement-learning-with-random-delays-1","title":"Reinforcement Learning with Random Delays","date":"2020-10-06","arxiv_id":"2010.02966","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-random-delays-1#ran","syntology_url":"https://syntology.ai/paper/2010.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02966"}},"official":{"repos":["rmst/rlrd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/exploration-in-approximate-hyper-state-space","slug":"exploration-in-approximate-hyper-state-space","title":"Exploration in Approximate Hyper-State Space for Meta Reinforcement Learning","date":"2020-10-02","arxiv_id":"2010.01062","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/exploration-in-approximate-hyper-state-space#ran","syntology_url":"https://syntology.ai/paper/2010.01062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01062"}},"official":{"repos":["lmzintgraf/hyperx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-fully-offline-meta-reinforcement","slug":"efficient-fully-offline-meta-reinforcement","title":"FOCAL: Efficient Fully-Offline Meta-Reinforcement Learning via Distance Metric Learning and Behavior Regularization","date":"2020-10-02","arxiv_id":"2010.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-fully-offline-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01112"}},"official":{"repos":["FOCAL-ICLR/FOCAL-ICLR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-social-reinforcement-learning","slug":"multi-agent-social-reinforcement-learning","title":"Emergent Social Learning via Multi-agent Reinforcement Learning","date":"2020-10-01","arxiv_id":"2010.00581","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-social-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2010.00581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.00581"}},"official":null}},{"url":"/paper/pettingzoo-gym-for-multi-agent-reinforcement","slug":"pettingzoo-gym-for-multi-agent-reinforcement","title":"PettingZoo: Gym for Multi-Agent Reinforcement Learning","date":"2020-09-30","arxiv_id":"2009.14471","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pettingzoo-gym-for-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.14471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.14471"}},"official":{"repos":["Farama-Foundation/PettingZoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decoupling-representation-learning-from","slug":"decoupling-representation-learning-from","title":"Decoupling Representation Learning from Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.08319","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-representation-learning-from#ran","syntology_url":"https://syntology.ai/paper/2009.08319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.08319"}},"official":{"repos":["astooke/rlpyt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynode-neural-ordinary-differential-equations","slug":"dynode-neural-ordinary-differential-equations","title":"DyNODE: Neural Ordinary Differential Equations for Dynamics Modeling in Continuous Control","date":"2020-09-09","arxiv_id":"2009.04278","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynode-neural-ordinary-differential-equations#ran","syntology_url":"https://syntology.ai/paper/2009.04278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.04278"}},"official":{"repos":["vmartinezalvarez/DyNODE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-automated-deep-reinforcement","slug":"sample-efficient-automated-deep-reinforcement","title":"Sample-Efficient Automated Deep Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01555","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-automated-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.01555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.01555"}},"official":{"repos":["automl/SEARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-off-policy-with-online-planning","slug":"learning-off-policy-with-online-planning","title":"Learning Off-Policy with Online Planning","date":"2020-08-23","arxiv_id":"2008.10066","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-off-policy-with-online-planning#ran","syntology_url":"https://syntology.ai/paper/2008.10066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.10066"}},"official":null}},{"url":"/paper/a-composable-specification-language-for-1","slug":"a-composable-specification-language-for-1","title":"A Composable Specification Language for Reinforcement Learning Tasks","date":"2020-08-21","arxiv_id":"2008.09293","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-composable-specification-language-for-1#ran","syntology_url":"https://syntology.ai/paper/2008.09293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09293"}},"official":{"repos":["keyshor/spectrl_tool"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-in-constrained","slug":"safe-reinforcement-learning-in-constrained","title":"Safe Reinforcement Learning in Constrained Markov Decision Processes","date":"2020-08-15","arxiv_id":"2008.06626","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-reinforcement-learning-in-constrained#ran","syntology_url":"https://syntology.ai/paper/2008.06626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06626"}},"official":{"repos":["akifumi-wachi-4/safe_near_optimal_mdp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-meta-reinforcement-learning-with","slug":"offline-meta-reinforcement-learning-with","title":"Offline Meta-Reinforcement Learning with Advantage Weighting","date":"2020-08-13","arxiv_id":"2008.06043","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2008.06043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06043"}},"official":{"repos":["eric-mitchell/macaw"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/trifinger-an-open-source-robot-for-learning","slug":"trifinger-an-open-source-robot-for-learning","title":"TriFinger: An Open-Source Robot for Learning Dexterity","date":"2020-08-08","arxiv_id":"2008.03596","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trifinger-an-open-source-robot-for-learning#ran","syntology_url":"https://syntology.ai/paper/2008.03596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.03596"}},"official":null}},{"url":"/paper/the-emergence-of-adversarial-communication-in","slug":"the-emergence-of-adversarial-communication-in","title":"The Emergence of Adversarial Communication in Multi-Agent Reinforcement Learning","date":"2020-08-06","arxiv_id":"2008.02616","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-emergence-of-adversarial-communication-in#ran","syntology_url":"https://syntology.ai/paper/2008.02616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02616"}},"official":{"repos":["proroklab/adversarial_comms"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-deep-reinforcement-learning-through","slug":"robust-deep-reinforcement-learning-through","title":"Robust Deep Reinforcement Learning through Adversarial Loss","date":"2020-08-05","arxiv_id":"2008.01976","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through#ran","syntology_url":"https://syntology.ai/paper/2008.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01976"}},"official":{"repos":["tuomaso/radial_rl","tuomaso/radial_rl_v2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/queueing-network-controls-via-deep","slug":"queueing-network-controls-via-deep","title":"Queueing Network Controls via Deep Reinforcement Learning","date":"2020-07-31","arxiv_id":"2008.01644","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/queueing-network-controls-via-deep#ran","syntology_url":"https://syntology.ai/paper/2008.01644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01644"}},"official":null}},{"url":"/paper/munchausen-reinforcement-learning","slug":"munchausen-reinforcement-learning","title":"Munchausen Reinforcement Learning","date":"2020-07-28","arxiv_id":"2007.14430","repositories_listed":6,"syntology":{"n":22,"n_ran":12,"n_constructed":11,"n_ran_checked":12,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/munchausen-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.14430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.14430"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/combining-deep-reinforcement-learning-and-1","slug":"combining-deep-reinforcement-learning-and-1","title":"Combining Deep Reinforcement Learning and Search for Imperfect-Information Games","date":"2020-07-27","arxiv_id":"2007.13544","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/combining-deep-reinforcement-learning-and-1#ran","syntology_url":"https://syntology.ai/paper/2007.13544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.13544"}},"official":{"repos":["facebookresearch/rebel"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-robust-optimization-for-imitation","slug":"bayesian-robust-optimization-for-imitation","title":"Bayesian Robust Optimization for Imitation Learning","date":"2020-07-24","arxiv_id":"2007.12315","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayesian-robust-optimization-for-imitation#ran","syntology_url":"https://syntology.ai/paper/2007.12315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12315"}},"official":{"repos":["dsbrown1331/broil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reinforcement-learning-with-3","slug":"distributional-reinforcement-learning-with-3","title":"Distributional Reinforcement Learning via Moment Matching","date":"2020-07-24","arxiv_id":"2007.12354","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distributional-reinforcement-learning-with-3#ran","syntology_url":"https://syntology.ai/paper/2007.12354","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12354"}},"official":{"repos":["thanhnguyentang/mmdrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/babyai-1-1","slug":"babyai-1-1","title":"BabyAI 1.1","date":"2020-07-24","arxiv_id":"2007.12770","repositories_listed":3,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/babyai-1-1#ran","syntology_url":"https://syntology.ai/paper/2007.12770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12770"}},"official":{"repos":["mila-iqia/babyai"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/active-mr-k-space-sampling-with-reinforcement","slug":"active-mr-k-space-sampling-with-reinforcement","title":"Active MR k-space Sampling with Reinforcement Learning","date":"2020-07-20","arxiv_id":"2007.10469","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/active-mr-k-space-sampling-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.10469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10469"}},"official":{"repos":["facebookresearch/active-mri-acquisition"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-good-batch-reinforcement-learning","slug":"provably-good-batch-reinforcement-learning","title":"Provably Good Batch Reinforcement Learning Without Great Exploration","date":"2020-07-16","arxiv_id":"2007.08202","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/provably-good-batch-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.08202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.08202"}},"official":null}},{"url":"/paper/implicit-distributional-reinforcement","slug":"implicit-distributional-reinforcement","title":"Implicit Distributional Reinforcement Learning","date":"2020-07-13","arxiv_id":"2007.06159","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/implicit-distributional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.06159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06159"}},"official":{"repos":["zhougroup/IDAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/data-efficient-reinforcement-learning-with-1","slug":"data-efficient-reinforcement-learning-with-1","title":"Data-Efficient Reinforcement Learning with Self-Predictive Representations","date":"2020-07-12","arxiv_id":"2007.05929","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/data-efficient-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2007.05929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05929"}},"official":{"repos":["mila-iqia/spr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/long-term-planning-with-deep-reinforcement","slug":"long-term-planning-with-deep-reinforcement","title":"Long-Term Planning with Deep Reinforcement Learning on Autonomous Drones","date":"2020-07-11","arxiv_id":"2007.05694","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/long-term-planning-with-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.05694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05694"}},"official":{"repos":["ugurkanates/NeurIRS2019DroneChallengeRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-sat-solvers-with-glue-variable","slug":"enhancing-sat-solvers-with-glue-variable","title":"Enhancing SAT solvers with glue variable predictions","date":"2020-07-06","arxiv_id":"2007.02559","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":2,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-sat-solvers-with-glue-variable#ran","syntology_url":"https://syntology.ai/paper/2007.02559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02559"}},"official":{"repos":["jesse-michael-han/neuro-cadical"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-data-augmentation-using","slug":"counterfactual-data-augmentation-using","title":"Counterfactual Data Augmentation using Locally Factored Dynamics","date":"2020-07-06","arxiv_id":"2007.02863","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-data-augmentation-using#ran","syntology_url":"https://syntology.ai/paper/2007.02863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02863"}},"official":null}},{"url":"/paper/discount-factor-as-a-regularizer-in","slug":"discount-factor-as-a-regularizer-in","title":"Discount Factor as a Regularizer in Reinforcement Learning","date":"2020-07-04","arxiv_id":"2007.02040","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discount-factor-as-a-regularizer-in#ran","syntology_url":"https://syntology.ai/paper/2007.02040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.02040"}},"official":{"repos":["ron-amit/Discount_as_Regularizer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mdp-homomorphic-networks-group-symmetries-in","slug":"mdp-homomorphic-networks-group-symmetries-in","title":"MDP Homomorphic Networks: Group Symmetries in Reinforcement Learning","date":"2020-06-30","arxiv_id":"2006.16908","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-homomorphic-networks-group-symmetries-in#ran","syntology_url":"https://syntology.ai/paper/2006.16908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16908"}},"official":{"repos":["ElisevanderPol/mdp-homomorphic-networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/off-dynamics-reinforcement-learning-training","slug":"off-dynamics-reinforcement-learning-training","title":"Off-Dynamics Reinforcement Learning: Training for Transfer with Domain Classifiers","date":"2020-06-24","arxiv_id":"2006.13916","repositories_listed":2,"syntology":{"n":23,"n_ran":21,"n_constructed":4,"n_ran_checked":15,"n_instrument":6,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":12,"n_pointer_only":7,"phrase":"21 ran (of which 4 constructed an object rather than computing a result; 15 with no instrument failure: 3 honoured, 0 violated, 12 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/off-dynamics-reinforcement-learning-training#ran","syntology_url":"https://syntology.ai/paper/2006.13916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13916"}},"official":null}},{"url":"/paper/automatic-data-augmentation-for","slug":"automatic-data-augmentation-for","title":"Automatic Data Augmentation for Generalization in Deep Reinforcement Learning","date":"2020-06-23","arxiv_id":"2006.12862","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automatic-data-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2006.12862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.12862"}},"official":{"repos":["rraileanu/auto-drac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/expert-supervised-reinforcement-learning-for","slug":"expert-supervised-reinforcement-learning-for","title":"Expert-Supervised Reinforcement Learning for Offline Policy Learning and Evaluation","date":"2020-06-23","arxiv_id":"2006.13189","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":2,"n_ran_checked":2,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":11,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/expert-supervised-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2006.13189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13189"}},"official":{"repos":["asonabend/ESRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/sample-factory-egocentric-3d-control-from","slug":"sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-factory-egocentric-3d-control-from#ran","syntology_url":"https://syntology.ai/paper/2006.11751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11751"}},"official":{"repos":["alex-petrenko/sample-factory"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-adjacency-constrained-subgoals-in","slug":"generating-adjacency-constrained-subgoals-in","title":"Generating Adjacency-Constrained Subgoals in Hierarchical Reinforcement Learning","date":"2020-06-20","arxiv_id":"2006.11485","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generating-adjacency-constrained-subgoals-in#ran","syntology_url":"https://syntology.ai/paper/2006.11485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11485"}},"official":{"repos":["trzhang0116/HRAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/accelerating-online-reinforcement-learning","slug":"accelerating-online-reinforcement-learning","title":"AWAC: Accelerating Online Reinforcement Learning with Offline Datasets","date":"2020-06-16","arxiv_id":"2006.09359","repositories_listed":6,"syntology":{"n":13,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":8,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/accelerating-online-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.09359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09359"}},"official":{"repos":["vitchyr/rlkit"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/opponent-modelling-with-local-information","slug":"opponent-modelling-with-local-information","title":"Agent Modelling under Partial Observability for Deep Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.09447","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/opponent-modelling-with-local-information#ran","syntology_url":"https://syntology.ai/paper/2006.09447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09447"}},"official":{"repos":["uoe-agents/LIAM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learn-to-effectively-explore-in-context-based","slug":"learn-to-effectively-explore-in-context-based","title":"MetaCURE: Meta Reinforcement Learning with Empowerment-Driven Exploration","date":"2020-06-15","arxiv_id":"2006.08170","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learn-to-effectively-explore-in-context-based#ran","syntology_url":"https://syntology.ai/paper/2006.08170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08170"}},"official":{"repos":["NagisaZj/MetaCURE-Public"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pipeline-psro-a-scalable-approach-for-finding","slug":"pipeline-psro-a-scalable-approach-for-finding","title":"Pipeline PSRO: A Scalable Approach for Finding Approximate Nash Equilibria in Large Games","date":"2020-06-15","arxiv_id":"2006.08555","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pipeline-psro-a-scalable-approach-for-finding#ran","syntology_url":"https://syntology.ai/paper/2006.08555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08555"}},"official":{"repos":["JBLanier/pipeline-psro"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comparative-evaluation-of-multi-agent-deep","slug":"comparative-evaluation-of-multi-agent-deep","title":"Benchmarking Multi-Agent Deep Reinforcement Learning Algorithms in Cooperative Tasks","date":"2020-06-14","arxiv_id":"2006.07869","repositories_listed":9,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comparative-evaluation-of-multi-agent-deep#ran","syntology_url":"https://syntology.ai/paper/2006.07869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07869"}},"official":{"repos":["uoe-agents/epymarl","uoe-agents/robotic-warehouse","uoe-agents/lb-foraging"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/torsionnet-a-reinforcement-learning-approach","slug":"torsionnet-a-reinforcement-learning-approach","title":"TorsionNet: A Reinforcement Learning Approach to Sequential Conformer Search","date":"2020-06-12","arxiv_id":"2006.07078","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/torsionnet-a-reinforcement-learning-approach#ran","syntology_url":"https://syntology.ai/paper/2006.07078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07078"}},"official":{"repos":["tarungog/torsionnet_paper_version"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/distributed-reinforcement-learning-in-multi","slug":"distributed-reinforcement-learning-in-multi","title":"Multi-Agent Reinforcement Learning in Stochastic Networked Systems","date":"2020-06-11","arxiv_id":"2006.06555","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distributed-reinforcement-learning-in-multi#ran","syntology_url":"https://syntology.ai/paper/2006.06555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06555"}},"official":null}},{"url":"/paper/constrained-episodic-reinforcement-learning","slug":"constrained-episodic-reinforcement-learning","title":"Constrained episodic reinforcement learning in concave-convex and knapsack settings","date":"2020-06-09","arxiv_id":"2006.05051","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/constrained-episodic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.05051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05051"}},"official":{"repos":["miryoosefi/ConRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-under-moral","slug":"reinforcement-learning-under-moral","title":"Reinforcement Learning Under Moral Uncertainty","date":"2020-06-08","arxiv_id":"2006.04734","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-under-moral#ran","syntology_url":"https://syntology.ai/paper/2006.04734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04734"}},"official":{"repos":["uber-research/normative-uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-q-learning-for-offline","slug":"conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","arxiv_id":"2006.04779","repositories_listed":18,"syntology":{"n":34,"n_ran":31,"n_constructed":5,"n_ran_checked":28,"n_instrument":3,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":26,"n_pointer_only":19,"phrase":"31 ran (of which 5 constructed an object rather than computing a result; 28 with no instrument failure: 2 honoured, 0 violated, 26 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2006.04779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04779"}},"official":{"repos":["aviralkumar2907/CQL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-for-multi-product","slug":"reinforcement-learning-for-multi-product","title":"Reinforcement Learning for Multi-Product Multi-Node Inventory Management in Supply Chains","date":"2020-06-07","arxiv_id":"2006.04037","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-for-multi-product#ran","syntology_url":"https://syntology.ai/paper/2006.04037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04037"}},"official":null}},{"url":"/paper/dual-policy-distillation","slug":"dual-policy-distillation","title":"Dual Policy Distillation","date":"2020-06-07","arxiv_id":"2006.04061","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/dual-policy-distillation#ran","syntology_url":"https://syntology.ai/paper/2006.04061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04061"}},"official":null}},{"url":"/paper/deployment-efficient-reinforcement-learning","slug":"deployment-efficient-reinforcement-learning","title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","date":"2020-06-05","arxiv_id":"2006.03647","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deployment-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03647"}},"official":{"repos":["matsuolab/BREMEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-transfer-for-reinforcement-learning","slug":"visual-transfer-for-reinforcement-learning","title":"Visual Transfer for Reinforcement Learning via Wasserstein Domain Confusion","date":"2020-06-04","arxiv_id":"2006.03465","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-transfer-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03465"}},"official":null}},{"url":"/paper/predicting-goal-directed-human-attention","slug":"predicting-goal-directed-human-attention","title":"Predicting Goal-directed Human Attention Using Inverse Reinforcement Learning","date":"2020-05-28","arxiv_id":"2005.14310","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/predicting-goal-directed-human-attention#ran","syntology_url":"https://syntology.ai/paper/2005.14310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14310"}},"official":{"repos":["cvlab-stonybrook/Scanpath_Prediction"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-state-dependent-exploration-for#ran","syntology_url":"https://syntology.ai/paper/2005.05719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05719"}},"official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/morel-model-based-offline-reinforcement","slug":"morel-model-based-offline-reinforcement","title":"MOReL : Model-Based Offline Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05951","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/morel-model-based-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2005.05951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05951"}},"official":null}},{"url":"/paper/planning-to-explore-via-self-supervised-world","slug":"planning-to-explore-via-self-supervised-world","title":"Planning to Explore via Self-Supervised World Models","date":"2020-05-12","arxiv_id":"2005.05960","repositories_listed":4,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/planning-to-explore-via-self-supervised-world#ran","syntology_url":"https://syntology.ai/paper/2005.05960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05960"}},"official":{"repos":["ramanans1/plan2explore"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/carl-controllable-agent-with-reinforcement","slug":"carl-controllable-agent-with-reinforcement","title":"CARL: Controllable Agent with Reinforcement Learning for Quadruped Locomotion","date":"2020-05-07","arxiv_id":"2005.03288","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/carl-controllable-agent-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2005.03288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03288"}},"official":{"repos":["inventec-ai-center/carl-siggraph2020"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/curious-hierarchical-actor-critic","slug":"curious-hierarchical-actor-critic","title":"Curious Hierarchical Actor-Critic Reinforcement Learning","date":"2020-05-07","arxiv_id":"2005.03420","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curious-hierarchical-actor-critic#ran","syntology_url":"https://syntology.ai/paper/2005.03420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03420"}},"official":{"repos":["knowledgetechnologyuhh/goal_conditioned_RL_baselines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-augmented-data","slug":"reinforcement-learning-with-augmented-data","title":"Reinforcement Learning with Augmented Data","date":"2020-04-30","arxiv_id":"2004.14990","repositories_listed":2,"syntology":{"n":24,"n_ran":21,"n_constructed":8,"n_ran_checked":9,"n_instrument":12,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":21,"phrase":"21 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 12 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-augmented-data#ran","syntology_url":"https://syntology.ai/paper/2004.14990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14990"}},"official":null}},{"url":"/paper/image-augmentation-is-all-you-need","slug":"image-augmentation-is-all-you-need","title":"Image Augmentation Is All You Need: Regularizing Deep Reinforcement Learning from Pixels","date":"2020-04-28","arxiv_id":"2004.13649","repositories_listed":4,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-augmentation-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2004.13649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.13649"}},"official":{"repos":["denisyarats/drq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/first-return-then-explore","slug":"first-return-then-explore","title":"First return, then explore","date":"2020-04-27","arxiv_id":"2004.12919","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/first-return-then-explore#ran","syntology_url":"https://syntology.ai/paper/2004.12919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.12919"}},"official":{"repos":["uber-research/go-explore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-real-world-robotic-skills-via","slug":"emergent-real-world-robotic-skills-via","title":"Emergent Real-World Robotic Skills via Unsupervised Off-Policy Reinforcement Learning","date":"2020-04-27","arxiv_id":"2004.12974","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergent-real-world-robotic-skills-via#ran","syntology_url":"https://syntology.ai/paper/2004.12974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.12974"}},"official":null}},{"url":"/paper/chip-placement-with-deep-reinforcement","slug":"chip-placement-with-deep-reinforcement","title":"Chip Placement with Deep Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10746","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chip-placement-with-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.10746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.10746"}},"official":null}},{"url":"/paper/energy-based-imitation-learning","slug":"energy-based-imitation-learning","title":"Energy-Based Imitation Learning","date":"2020-04-20","arxiv_id":"2004.09395","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/energy-based-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2004.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09395"}},"official":{"repos":["apexrl/EBIL-torch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-reinforcement-learning-with-multi","slug":"continual-reinforcement-learning-with-multi","title":"Continual Reinforcement Learning with Multi-Timescale Replay","date":"2020-04-16","arxiv_id":"2004.07530","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/continual-reinforcement-learning-with-multi#ran","syntology_url":"https://syntology.ai/paper/2004.07530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07530"}},"official":{"repos":["ChristosKap/multi_timescale_replay"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-game-theoretic-framework-for-model-based","slug":"a-game-theoretic-framework-for-model-based","title":"A Game Theoretic Framework for Model Based Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07804","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-game-theoretic-framework-for-model-based#ran","syntology_url":"https://syntology.ai/paper/2004.07804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07804"}},"official":null}},{"url":"/paper/datasets-for-data-driven-reinforcement","slug":"datasets-for-data-driven-reinforcement","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07219","repositories_listed":7,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/datasets-for-data-driven-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.07219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07219"}},"official":{"repos":["rail-berkeley/d4rl","rail-berkeley/offline_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/curl-contrastive-unsupervised-representations","slug":"curl-contrastive-unsupervised-representations","title":"CURL: Contrastive Unsupervised Representations for Reinforcement Learning","date":"2020-04-08","arxiv_id":"2004.04136","repositories_listed":7,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curl-contrastive-unsupervised-representations#ran","syntology_url":"https://syntology.ai/paper/2004.04136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.04136"}},"official":{"repos":["MishaLaskin/curl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"cce4392dbd201ea02a5888a6ca11c80d4a0a541db88c51f1530d9dbfbb152ca9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}