{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/ran/8","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":8,"pages_in_order":12,"rows_per_page":100,"rows":[701,800],"of":1165,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2/papers/ran/1","prev":"/task/reinforcement-learning-2/papers/ran/7","next":"/task/reinforcement-learning-2/papers/ran/9","papers":[{"url":"/paper/learning-markov-state-abstractions-for-deep","slug":"learning-markov-state-abstractions-for-deep","title":"Learning Markov State Abstractions for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04379","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-markov-state-abstractions-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04379"}},"official":{"repos":["camall3n/markov-state-abstractions"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/task-driven-semantic-coding-via-reinforcement","slug":"task-driven-semantic-coding-via-reinforcement","title":"Task-driven Semantic Coding via Reinforcement Learning","date":"2021-06-07","arxiv_id":"2106.03511","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-driven-semantic-coding-via-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03511"}},"official":{"repos":["USTC-IMCL/Task-driven-Semantic-Coding-via-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/verifiable-and-compositional-reinforcement","slug":"verifiable-and-compositional-reinforcement","title":"Verifiable and Compositional Reinforcement Learning Systems","date":"2021-06-07","arxiv_id":"2106.05864","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/verifiable-and-compositional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.05864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05864"}},"official":{"repos":["cyrusneary/verifiable-compositional-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/control-oriented-model-based-reinforcement","slug":"control-oriented-model-based-reinforcement","title":"Control-Oriented Model-Based Reinforcement Learning with Implicit Differentiation","date":"2021-06-06","arxiv_id":"2106.03273","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/control-oriented-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03273"}},"official":{"repos":["evgenii-nikishin/omd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/same-state-different-task-continual","slug":"same-state-different-task-continual","title":"Same State, Different Task: Continual Reinforcement Learning without Interference","date":"2021-06-05","arxiv_id":"2106.02940","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/same-state-different-task-continual#ran","syntology_url":"https://syntology.ai/paper/2106.02940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02940"}},"official":{"repos":["skezle/owl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/malib-a-parallel-framework-for-population","slug":"malib-a-parallel-framework-for-population","title":"MALib: A Parallel Framework for Population-based Multi-agent Reinforcement Learning","date":"2021-06-05","arxiv_id":"2106.07551","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/malib-a-parallel-framework-for-population#ran","syntology_url":"https://syntology.ai/paper/2106.07551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07551"}},"official":{"repos":["sjtu-marl/malib"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/optimization-based-algebraic-multigrid","slug":"optimization-based-algebraic-multigrid","title":"Optimization-Based Algebraic Multigrid Coarsening Using Reinforcement Learning","date":"2021-06-03","arxiv_id":"2106.01854","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimization-based-algebraic-multigrid#ran","syntology_url":"https://syntology.ai/paper/2106.01854","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01854"}},"official":{"repos":["compdyn/rl_grid_coarsen"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-as-one-big-sequence","slug":"reinforcement-learning-as-one-big-sequence","title":"Offline Reinforcement Learning as One Big Sequence Modeling Problem","date":"2021-06-03","arxiv_id":"2106.02039","repositories_listed":2,"syntology":{"n":28,"n_ran":23,"n_constructed":0,"n_ran_checked":22,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":21,"n_pointer_only":1,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 0 violated, 21 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/reinforcement-learning-as-one-big-sequence#ran","syntology_url":"https://syntology.ai/paper/2106.02039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02039"}},"official":{"repos":["JannerM/trajectory-transformer"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-consciousness-inspired-planning-agent-for","slug":"a-consciousness-inspired-planning-agent-for","title":"A Consciousness-Inspired Planning Agent for Model-Based Reinforcement Learning","date":"2021-06-03","arxiv_id":"2106.02097","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-consciousness-inspired-planning-agent-for#ran","syntology_url":"https://syntology.ai/paper/2106.02097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02097"}},"official":{"repos":["mila-iqia/conscious-planning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/mico-learning-improved-representations-via","slug":"mico-learning-improved-representations-via","title":"MICo: Improved representations via sampling-based state similarity for Markov decision processes","date":"2021-06-03","arxiv_id":"2106.08229","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mico-learning-improved-representations-via#ran","syntology_url":"https://syntology.ai/paper/2106.08229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08229"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/decision-transformer-reinforcement-learning","slug":"decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","arxiv_id":"2106.01345","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_constructed":10,"n_ran_checked":13,"n_instrument":4,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":8,"phrase":"17 ran (of which 10 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/decision-transformer-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.01345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01345"}},"official":{"repos":["kzl/decision-transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/improving-generalization-in-meta-rl-with","slug":"improving-generalization-in-meta-rl-with","title":"Improving Generalization in Meta-RL with Imaginary Tasks from Latent Dynamics Mixture","date":"2021-05-28","arxiv_id":"2105.13524","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/improving-generalization-in-meta-rl-with#ran","syntology_url":"https://syntology.ai/paper/2105.13524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13524"}},"official":{"repos":["suyoung-lee/ldm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/towards-mental-time-travel-a-hierarchical","slug":"towards-mental-time-travel-a-hierarchical","title":"Towards mental time travel: a hierarchical memory for reinforcement learning agents","date":"2021-05-28","arxiv_id":"2105.14039","repositories_listed":5,"syntology":{"n":6,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/towards-mental-time-travel-a-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2105.14039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14039"}},"official":{"repos":["deepmind/deepmind-research","deepmind/dm_fast_mapping"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/objective-robustness-in-deep-reinforcement","slug":"objective-robustness-in-deep-reinforcement","title":"Goal Misgeneralization in Deep Reinforcement Learning","date":"2021-05-28","arxiv_id":"2105.14111","repositories_listed":4,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/objective-robustness-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.14111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14111"}},"official":{"repos":["JacobPfau/procgenAISC","jbkjr/train-procgen-pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/androidenv-a-reinforcement-learning-platform","slug":"androidenv-a-reinforcement-learning-platform","title":"AndroidEnv: A Reinforcement Learning Platform for Android","date":"2021-05-27","arxiv_id":"2105.13231","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/androidenv-a-reinforcement-learning-platform#ran","syntology_url":"https://syntology.ai/paper/2105.13231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13231"}},"official":{"repos":["deepmind/android_env"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/adversarial-intrinsic-motivation-for","slug":"adversarial-intrinsic-motivation-for","title":"Adversarial Intrinsic Motivation for Reinforcement Learning","date":"2021-05-27","arxiv_id":"2105.13345","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/adversarial-intrinsic-motivation-for#ran","syntology_url":"https://syntology.ai/paper/2105.13345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13345"}},"official":{"repos":["iDurugkar/adversarial-intrinsic-motivation"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/feasible-actor-critic-constrained","slug":"feasible-actor-critic-constrained","title":"Feasible Actor-Critic: Constrained Reinforcement Learning for Ensuring Statewise Safety","date":"2021-05-22","arxiv_id":"2105.10682","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/feasible-actor-critic-constrained#ran","syntology_url":"https://syntology.ai/paper/2105.10682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10682"}},"official":{"repos":["mahaitongdae/Feasible-Actor-Critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/minimum-delay-adaptation-in-non-stationary","slug":"minimum-delay-adaptation-in-non-stationary","title":"Minimum-Delay Adaptation in Non-Stationary Reinforcement Learning via Online High-Confidence Change-Point Detection","date":"2021-05-20","arxiv_id":"2105.09452","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/minimum-delay-adaptation-in-non-stationary#ran","syntology_url":"https://syntology.ai/paper/2105.09452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.09452"}},"official":{"repos":["LucasAlegre/mbcd"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/coach-player-multi-agent-reinforcement","slug":"coach-player-multi-agent-reinforcement","title":"Coach-Player Multi-Agent Reinforcement Learning for Dynamic Team Composition","date":"2021-05-18","arxiv_id":"2105.08692","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coach-player-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.08692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08692"}},"official":{"repos":["cranial-xix/marl-copa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-learning-based-deep-reinforcement","slug":"meta-learning-based-deep-reinforcement","title":"Meta-Learning-Based Deep Reinforcement Learning for Multiobjective Optimization Problems","date":"2021-05-06","arxiv_id":"2105.02741","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meta-learning-based-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.02741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.02741"}},"official":{"repos":["zhangzizhen/ml-dam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-to-reward-progressivity-via-spectral-1","slug":"adapting-to-reward-progressivity-via-spectral-1","title":"Adapting to Reward Progressivity via Spectral Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14138","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/adapting-to-reward-progressivity-via-spectral-1#ran","syntology_url":"https://syntology.ai/paper/2104.14138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.14138"}},"official":{"repos":["mchldann/SpectralDQN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-navigation-with-spatial-attention","slug":"visual-navigation-with-spatial-attention","title":"Visual Navigation with Spatial Attention","date":"2021-04-20","arxiv_id":"2104.09807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-navigation-with-spatial-attention#ran","syntology_url":"https://syntology.ai/paper/2104.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09807"}},"official":null}},{"url":"/paper/probabilistic-mixture-of-experts-for-1","slug":"probabilistic-mixture-of-experts-for-1","title":"Probabilistic Mixture-of-Experts for Efficient Deep Reinforcement Learning","date":"2021-04-19","arxiv_id":"2104.09122","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/probabilistic-mixture-of-experts-for-1#ran","syntology_url":"https://syntology.ai/paper/2104.09122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.09122"}},"official":{"repos":["JieRen98/rlkit-pmoe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalising-discrete-action-spaces-with","slug":"generalising-discrete-action-spaces-with","title":"Generalising Discrete Action Spaces with Conditional Action Trees","date":"2021-04-15","arxiv_id":"2104.07294","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generalising-discrete-action-spaces-with#ran","syntology_url":"https://syntology.ai/paper/2104.07294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.07294"}},"official":{"repos":["Bam4d/conditional-action-trees"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/podracer-architectures-for-scalable","slug":"podracer-architectures-for-scalable","title":"Podracer architectures for scalable Reinforcement Learning","date":"2021-04-13","arxiv_id":"2104.06272","repositories_listed":3,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/podracer-architectures-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2104.06272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06272"}},"official":null}},{"url":"/paper/character-controllers-using-motion-vaes","slug":"character-controllers-using-motion-vaes","title":"Character Controllers Using Motion VAEs","date":"2021-03-26","arxiv_id":"2103.14274","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/character-controllers-using-motion-vaes#ran","syntology_url":"https://syntology.ai/paper/2103.14274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.14274"}},"official":{"repos":["electronicarts/character-motion-vaes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-predictive-actor-critic-accelerating","slug":"model-predictive-actor-critic-accelerating","title":"Model Predictive Actor-Critic: Accelerating Robot Skill Acquisition with Deep Reinforcement Learning","date":"2021-03-25","arxiv_id":"2103.13842","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-predictive-actor-critic-accelerating#ran","syntology_url":"https://syntology.ai/paper/2103.13842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13842"}},"official":{"repos":["dnandha/mopac"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-compositional-reinforcement-learning-1","slug":"solving-compositional-reinforcement-learning-1","title":"Solving Compositional Reinforcement Learning Problems via Task Reduction","date":"2021-03-13","arxiv_id":"2103.07607","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/solving-compositional-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2103.07607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.07607"}},"official":{"repos":["IrisLi17/self-imitation-via-reduction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizable-episodic-memory-for-deep","slug":"generalizable-episodic-memory-for-deep","title":"Generalizable Episodic Memory for Deep Reinforcement Learning","date":"2021-03-11","arxiv_id":"2103.06469","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalizable-episodic-memory-for-deep#ran","syntology_url":"https://syntology.ai/paper/2103.06469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.06469"}},"official":{"repos":["MouseHu/GEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-safe-reinforcement-learning-using","slug":"model-based-safe-reinforcement-learning-using","title":"Model-based Constrained Reinforcement Learning using Generalized Control Barrier Function","date":"2021-03-02","arxiv_id":"2103.01556","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-safe-reinforcement-learning-using#ran","syntology_url":"https://syntology.ai/paper/2103.01556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.01556"}},"official":{"repos":["mahaitongdae/safe_exp_env"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-surprising-effectiveness-of-mappo-in","slug":"the-surprising-effectiveness-of-mappo-in","title":"The Surprising Effectiveness of PPO in Cooperative, Multi-Agent Games","date":"2021-03-02","arxiv_id":"2103.01955","repositories_listed":19,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-surprising-effectiveness-of-mappo-in#ran","syntology_url":"https://syntology.ai/paper/2103.01955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.01955"}},"official":{"repos":["marlbenchmark/on-policy"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-deep-reinforcement-learning-via-multi","slug":"robust-deep-reinforcement-learning-via-multi","title":"DRIBO: Robust Deep Reinforcement Learning via Multi-View Information Bottleneck","date":"2021-02-26","arxiv_id":"2102.13268","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-via-multi#ran","syntology_url":"https://syntology.ai/paper/2102.13268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.13268"}},"official":{"repos":["BU-DEPEND-Lab/DRIBO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-probabilistic-interpretation-of-self-paced","slug":"a-probabilistic-interpretation-of-self-paced","title":"A Probabilistic Interpretation of Self-Paced Learning with Applications to Reinforcement Learning","date":"2021-02-25","arxiv_id":"2102.13176","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-probabilistic-interpretation-of-self-paced#ran","syntology_url":"https://syntology.ai/paper/2102.13176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.13176"}},"official":{"repos":["psclklnk/self-paced-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/memory-based-deep-reinforcement-learning-for-1","slug":"memory-based-deep-reinforcement-learning-for-1","title":"Memory-based Deep Reinforcement Learning for POMDPs","date":"2021-02-24","arxiv_id":"2102.12344","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/memory-based-deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2102.12344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.12344"}},"official":{"repos":["LinghengMeng/LSTM-TD3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/information-directed-reward-learning-for","slug":"information-directed-reward-learning-for","title":"Information Directed Reward Learning for Reinforcement Learning","date":"2021-02-24","arxiv_id":"2102.12466","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/information-directed-reward-learning-for#ran","syntology_url":"https://syntology.ai/paper/2102.12466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.12466"}},"official":{"repos":["david-lindner/idrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/program-synthesis-guided-reinforcement","slug":"program-synthesis-guided-reinforcement","title":"Program Synthesis Guided Reinforcement Learning for Partially Observed Environments","date":"2021-02-22","arxiv_id":"2102.11137","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/program-synthesis-guided-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2102.11137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.11137"}},"official":{"repos":["yycdavid/program-synthesis-guided-rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/explore-the-context-optimal-data-collection","slug":"explore-the-context-optimal-data-collection","title":"Explore the Context: Optimal Data Collection for Context-Conditional Dynamics Models","date":"2021-02-22","arxiv_id":"2102.11394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explore-the-context-optimal-data-collection#ran","syntology_url":"https://syntology.ai/paper/2102.11394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.11394"}},"official":null}},{"url":"/paper/continuous-doubly-constrained-batch","slug":"continuous-doubly-constrained-batch","title":"Continuous Doubly Constrained Batch Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09225","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/continuous-doubly-constrained-batch#ran","syntology_url":"https://syntology.ai/paper/2102.09225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.09225"}},"official":{"repos":["amazon-research/cdc-batch-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-rational-networks#ran","syntology_url":"https://syntology.ai/paper/2102.09407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.09407"}},"official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-multi-agent-reinforcement-learning","slug":"scaling-multi-agent-reinforcement-learning","title":"Scaling Multi-Agent Reinforcement Learning with Selective Parameter Sharing","date":"2021-02-15","arxiv_id":"2102.07475","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/scaling-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.07475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07475"}},"official":{"repos":["uoe-agents/seps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/domain-adaptation-in-reinforcement-learning","slug":"domain-adaptation-in-reinforcement-learning","title":"Domain Adaptation In Reinforcement Learning Via Latent Unified State Representation","date":"2021-02-10","arxiv_id":"2102.05714","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/domain-adaptation-in-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.05714"}},"official":{"repos":["KarlXing/LUSR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameter-tricks-in-multi-agent","slug":"hyperparameter-tricks-in-multi-agent","title":"Rethinking the Implementation Matters in Cooperative Multi-Agent Reinforcement Learning","date":"2021-02-06","arxiv_id":"2102.03479","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyperparameter-tricks-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2102.03479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03479"}},"official":{"repos":["hijkzzz/pymarl2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/gnn-rl-compression-topology-aware-network","slug":"gnn-rl-compression-topology-aware-network","title":"Topology-Aware Network Pruning using Multi-stage Graph Embedding and Reinforcement Learning","date":"2021-02-05","arxiv_id":"2102.03214","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gnn-rl-compression-topology-aware-network#ran","syntology_url":"https://syntology.ai/paper/2102.03214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03214"}},"official":{"repos":["yusx-swapp/gnn-rl-model-compression"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-with","slug":"multi-agent-reinforcement-learning-with","title":"Multi-Agent Reinforcement Learning with Temporal Logic Specifications","date":"2021-02-01","arxiv_id":"2102.00582","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.00582"}},"official":{"repos":["lrhammond/almanac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-state-explanations-for","slug":"counterfactual-state-explanations-for","title":"Counterfactual State Explanations for Reinforcement Learning Agents via Generative Deep Learning","date":"2021-01-29","arxiv_id":"2101.12446","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-state-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2101.12446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.12446"}},"official":{"repos":["mattolson93/counterfactual-state-explanations"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-on-state-1","slug":"robust-reinforcement-learning-on-state-1","title":"Robust Reinforcement Learning on State Observations with Learned Optimal Adversary","date":"2021-01-21","arxiv_id":"2101.08452","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/robust-reinforcement-learning-on-state-1#ran","syntology_url":"https://syntology.ai/paper/2101.08452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08452"}},"official":{"repos":["huanzhang12/ATLA_robust_RL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/solving-common-payoff-games-with-approximate","slug":"solving-common-payoff-games-with-approximate","title":"Solving Common-Payoff Games with Approximate Policy Iteration","date":"2021-01-11","arxiv_id":"2101.04237","repositories_listed":2,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/solving-common-payoff-games-with-approximate#ran","syntology_url":"https://syntology.ai/paper/2101.04237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.04237"}},"official":{"repos":["ssokota/capi","ssokota/tiny-hanabi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-latent-flow-1","slug":"reinforcement-learning-with-latent-flow-1","title":"Reinforcement Learning with Latent Flow","date":"2021-01-06","arxiv_id":"2101.01857","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-latent-flow-1#ran","syntology_url":"https://syntology.ai/paper/2101.01857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.01857"}},"official":{"repos":["WendyShang/flare","WendyShang/dqn_zoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/augmenting-policy-learning-with-routines","slug":"augmenting-policy-learning-with-routines","title":"Augmenting Policy Learning with Routines Discovered from a Single Demonstration","date":"2020-12-23","arxiv_id":"2012.12469","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-policy-learning-with-routines#ran","syntology_url":"https://syntology.ai/paper/2012.12469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.12469"}},"official":{"repos":["sjtuytc/-AAAI21-RoutineAugmentedPolicyLearning-RAPL-","sjtuytc/AAAI21-RoutineAugmentedPolicyLearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-decoder-attention-model-with-embedding","slug":"multi-decoder-attention-model-with-embedding","title":"Multi-Decoder Attention Model with Embedding Glimpse for Solving Vehicle Routing Problems","date":"2020-12-19","arxiv_id":"2012.10638","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-decoder-attention-model-with-embedding#ran","syntology_url":"https://syntology.ai/paper/2012.10638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.10638"}},"official":{"repos":["liangxinedu/MDAM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/high-throughput-synchronous-deep-rl-1","slug":"high-throughput-synchronous-deep-rl-1","title":"High-Throughput Synchronous Deep RL","date":"2020-12-17","arxiv_id":"2012.09849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/high-throughput-synchronous-deep-rl-1#ran","syntology_url":"https://syntology.ai/paper/2012.09849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09849"}},"official":{"repos":["IouJenLiu/HTS-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-of-graph-matching","slug":"deep-reinforcement-learning-of-graph-matching","title":"Revocable Deep Reinforcement Learning with Affinity Regularization for Outlier-Robust Graph Matching","date":"2020-12-16","arxiv_id":"2012.08950","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-of-graph-matching#ran","syntology_url":"https://syntology.ai/paper/2012.08950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.08950"}},"official":{"repos":["thinklab-sjtu/rgm","Thinklab-SJTU/awesome-ml4co"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-constrained-reinforcement-learning","slug":"inverse-constrained-reinforcement-learning","title":"Inverse Constrained Reinforcement Learning","date":"2020-11-19","arxiv_id":"2011.09999","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/inverse-constrained-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2011.09999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.09999"}},"official":{"repos":["shehryar-malik/icrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-associative-inference-using-fast-1","slug":"learning-associative-inference-using-fast-1","title":"Learning Associative Inference Using Fast Weight Memory","date":"2020-11-16","arxiv_id":"2011.07831","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-associative-inference-using-fast-1#ran","syntology_url":"https://syntology.ai/paper/2011.07831","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07831"}},"official":{"repos":["ischlag/Fast-Weight-Memory-public"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-reinforcement-learning-policies-for","slug":"scalable-reinforcement-learning-policies-for","title":"Scalable Reinforcement Learning Policies for Multi-Agent Control","date":"2020-11-16","arxiv_id":"2011.08055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-reinforcement-learning-policies-for#ran","syntology_url":"https://syntology.ai/paper/2011.08055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.08055"}},"official":{"repos":["christopher-hsu/scalableMARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tonic-a-deep-reinforcement-learning-library#ran","syntology_url":"https://syntology.ai/paper/2011.07537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07537"}},"official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-to-new-actions-in-1","slug":"generalization-to-new-actions-in-1","title":"Generalization to New Actions in Reinforcement Learning","date":"2020-11-03","arxiv_id":"2011.01928","repositories_listed":2,"syntology":{"n":21,"n_ran":18,"n_constructed":14,"n_ran_checked":16,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":14,"n_pointer_only":20,"phrase":"18 ran (of which 14 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/generalization-to-new-actions-in-1#ran","syntology_url":"https://syntology.ai/paper/2011.01928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01928"}},"official":{"repos":["clvrai/new-actions-rl","clvrai/create"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":14,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/instance-based-generalization-in","slug":"instance-based-generalization-in","title":"Instance based Generalization in Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01089","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instance-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2011.01089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01089"}},"official":{"repos":["MartinBertran/InstanceAgnosticPolicyEnsembles"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ask-your-humans-using-human-instructions-to-1","slug":"ask-your-humans-using-human-instructions-to-1","title":"Ask Your Humans: Using Human Instructions to Improve Generalization in Reinforcement Learning","date":"2020-11-01","arxiv_id":"2011.00517","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ask-your-humans-using-human-instructions-to-1#ran","syntology_url":"https://syntology.ai/paper/2011.00517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00517"}},"official":{"repos":["valeriechen/ask-your-humans"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-overview-of-multi-agent-reinforcement","slug":"an-overview-of-multi-agent-reinforcement","title":"Game-Theoretic Multiagent Reinforcement Learning","date":"2020-11-01","arxiv_id":"2011.00583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-overview-of-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2011.00583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00583"}},"official":null}},{"url":"/paper/a-policy-gradient-algorithm-for-learning-to-1","slug":"a-policy-gradient-algorithm-for-learning-to-1","title":"A Policy Gradient Algorithm for Learning to Learn in Multiagent Reinforcement Learning","date":"2020-10-31","arxiv_id":"2011.00382","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-policy-gradient-algorithm-for-learning-to-1#ran","syntology_url":"https://syntology.ai/paper/2011.00382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00382"}},"official":{"repos":["dkkim93/meta-mapg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/pomo-policy-optimization-with-multiple-optima","slug":"pomo-policy-optimization-with-multiple-optima","title":"POMO: Policy Optimization with Multiple Optima for Reinforcement Learning","date":"2020-10-30","arxiv_id":"2010.16011","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/pomo-policy-optimization-with-multiple-optima#ran","syntology_url":"https://syntology.ai/paper/2010.16011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.16011"}},"official":{"repos":["yd-kwon/POMO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/recovery-rl-safe-reinforcement-learning-with","slug":"recovery-rl-safe-reinforcement-learning-with","title":"Recovery RL: Safe Reinforcement Learning with Learned Recovery Zones","date":"2020-10-29","arxiv_id":"2010.15920","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recovery-rl-safe-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2010.15920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.15920"}},"official":null}},{"url":"/paper/implicit-under-parameterization-inhibits-data-1","slug":"implicit-under-parameterization-inhibits-data-1","title":"Implicit Under-Parameterization Inhibits Data-Efficient Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-under-parameterization-inhibits-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.14498"}},"official":null}},{"url":"/paper/trajectory-wise-multiple-choice-learning-for","slug":"trajectory-wise-multiple-choice-learning-for","title":"Trajectory-wise Multiple Choice Learning for Dynamics Generalization in Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13303","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-wise-multiple-choice-learning-for#ran","syntology_url":"https://syntology.ai/paper/2010.13303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13303"}},"official":{"repos":["younggyoseo/trajectory_mcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-imagination-and-reality-for-model","slug":"bridging-imagination-and-reality-for-model","title":"Bridging Imagination and Reality for Model-Based Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12142","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-imagination-and-reality-for-model#ran","syntology_url":"https://syntology.ai/paper/2010.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12142"}},"official":{"repos":["Mehooz/BIRD_code"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/learning-to-dispatch-for-job-shop-scheduling","slug":"learning-to-dispatch-for-job-shop-scheduling","title":"Learning to Dispatch for Job Shop Scheduling via Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12367","repositories_listed":4,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-dispatch-for-job-shop-scheduling#ran","syntology_url":"https://syntology.ai/paper/2010.12367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12367"}},"official":{"repos":["zcajiayin/L2D"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-reinforcement-learning-with-stacked","slug":"deep-reinforcement-learning-with-stacked","title":"Deep Reinforcement Learning with Stacked Hierarchical Attention for Text-based Games","date":"2020-10-22","arxiv_id":"2010.11655","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-with-stacked#ran","syntology_url":"https://syntology.ai/paper/2010.11655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11655"}},"official":{"repos":["YunqiuXu/SHA-KG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/drift-detection-in-episodic-data-detect-when-1","slug":"drift-detection-in-episodic-data-detect-when-1","title":"Detecting Rewards Deterioration in Episodic Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11660","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drift-detection-in-episodic-data-detect-when-1#ran","syntology_url":"https://syntology.ai/paper/2010.11660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11660"}},"official":{"repos":["ido90/Rewards-Deterioration-Detection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-optimization-of","slug":"reinforcement-learning-for-optimization-of","title":"Reinforcement Learning for Optimization of COVID-19 Mitigation policies","date":"2020-10-20","arxiv_id":"2010.10560","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-for-optimization-of#ran","syntology_url":"https://syntology.ai/paper/2010.10560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10560"}},"official":{"repos":["SonyAI/PandemicSimulator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/d2rl-deep-dense-architectures-in-1","slug":"d2rl-deep-dense-architectures-in-1","title":"D2RL: Deep Dense Architectures in Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09163","repositories_listed":4,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/d2rl-deep-dense-architectures-in-1#ran","syntology_url":"https://syntology.ai/paper/2010.09163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09163"}},"official":{"repos":["pairlab/d2rl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","listed","official"]}}},{"url":"/paper/knowledge-guided-open-attribute-value","slug":"knowledge-guided-open-attribute-value","title":"Knowledge-guided Open Attribute Value Extraction with Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09189","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/knowledge-guided-open-attribute-value#ran","syntology_url":"https://syntology.ai/paper/2010.09189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09189"}},"official":{"repos":["yeliu0930/Knowledge-guided-Open-Attribute-Value-Extraction-with-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-policy-optimization-with","slug":"model-based-policy-optimization-with","title":"Model-based Policy Optimization with Unsupervised Model Adaptation","date":"2020-10-19","arxiv_id":"2010.09546","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/model-based-policy-optimization-with#ran","syntology_url":"https://syntology.ai/paper/2010.09546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09546"}},"official":{"repos":["RockySJ/ampo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-with-population","slug":"deep-reinforcement-learning-with-population","title":"Deep Reinforcement Learning with Population-Coded Spiking Neural Network for Continuous Control","date":"2020-10-19","arxiv_id":"2010.09635","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-with-population#ran","syntology_url":"https://syntology.ai/paper/2010.09635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09635"}},"official":{"repos":["combra-lab/pop-spiking-deep-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/smarts-scalable-multi-agent-reinforcement","slug":"smarts-scalable-multi-agent-reinforcement","title":"SMARTS: Scalable Multi-Agent Reinforcement Learning Training School for Autonomous Driving","date":"2020-10-19","arxiv_id":"2010.09776","repositories_listed":5,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/smarts-scalable-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.09776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09776"}},"official":{"repos":["huawei-noah/SMARTS"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/hyperparameter-auto-tuning-in-self-supervised","slug":"hyperparameter-auto-tuning-in-self-supervised","title":"Hyperparameter Auto-tuning in Self-Supervised Robotic Learning","date":"2020-10-16","arxiv_id":"2010.08252","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hyperparameter-auto-tuning-in-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.08252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.08252"}},"official":{"repos":["birlrobotics/rlkit_autotune"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/masked-contrastive-representation-learning","slug":"masked-contrastive-representation-learning","title":"Masked Contrastive Representation Learning for Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.07470","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/masked-contrastive-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2010.07470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07470"}},"official":{"repos":["teslacool/m-curl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with-1","slug":"multi-task-deep-reinforcement-learning-with-1","title":"Knowledge Transfer in Multi-Task Deep Reinforcement Learning for Continuous Control","date":"2020-10-15","arxiv_id":"2010.07494","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2010.07494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07494"}},"official":null}},{"url":"/paper/a-game-theoretic-analysis-of-networked-system","slug":"a-game-theoretic-analysis-of-networked-system","title":"A game-theoretic analysis of networked system control for common-pool resource management using multi-agent reinforcement learning","date":"2020-10-15","arxiv_id":"2010.07777","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-game-theoretic-analysis-of-networked-system#ran","syntology_url":"https://syntology.ai/paper/2010.07777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07777"}},"official":{"repos":["instadeepai/EGTA-NMARL"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-trust-region-policy-optimization","slug":"multi-agent-trust-region-policy-optimization","title":"Multi-Agent Trust Region Policy Optimization","date":"2020-10-15","arxiv_id":"2010.07916","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":7,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-trust-region-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2010.07916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07916"}},"official":null}},{"url":"/paper/human-centric-dialog-training-via-offline","slug":"human-centric-dialog-training-via-offline","title":"Human-centric Dialog Training via Offline Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05848","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-dialog-training-via-offline#ran","syntology_url":"https://syntology.ai/paper/2010.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05848"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/contrastive-explanations-for-reinforcement-2","slug":"contrastive-explanations-for-reinforcement-2","title":"Contrastive Explanations for Reinforcement Learning via Embedded Self Predictions","date":"2020-10-11","arxiv_id":"2010.05180","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/contrastive-explanations-for-reinforcement-2#ran","syntology_url":"https://syntology.ai/paper/2010.05180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05180"}},"official":{"repos":["SuerpX/Embedded-Self-Predictions"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diverse-exploration-via-infomax-options-1","slug":"diverse-exploration-via-infomax-options-1","title":"Learning Diverse Options via InfoMax Termination Critic","date":"2020-10-06","arxiv_id":"2010.02756","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diverse-exploration-via-infomax-options-1#ran","syntology_url":"https://syntology.ai/paper/2010.02756","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02756"}},"official":{"repos":["kngwyu/infomax-option-critic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-random-delays-1","slug":"reinforcement-learning-with-random-delays-1","title":"Reinforcement Learning with Random Delays","date":"2020-10-06","arxiv_id":"2010.02966","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-random-delays-1#ran","syntology_url":"https://syntology.ai/paper/2010.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02966"}},"official":{"repos":["rmst/rlrd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/exploration-in-approximate-hyper-state-space","slug":"exploration-in-approximate-hyper-state-space","title":"Exploration in Approximate Hyper-State Space for Meta Reinforcement Learning","date":"2020-10-02","arxiv_id":"2010.01062","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/exploration-in-approximate-hyper-state-space#ran","syntology_url":"https://syntology.ai/paper/2010.01062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01062"}},"official":{"repos":["lmzintgraf/hyperx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-fully-offline-meta-reinforcement","slug":"efficient-fully-offline-meta-reinforcement","title":"FOCAL: Efficient Fully-Offline Meta-Reinforcement Learning via Distance Metric Learning and Behavior Regularization","date":"2020-10-02","arxiv_id":"2010.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-fully-offline-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01112"}},"official":{"repos":["FOCAL-ICLR/FOCAL-ICLR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-social-reinforcement-learning","slug":"multi-agent-social-reinforcement-learning","title":"Emergent Social Learning via Multi-agent Reinforcement Learning","date":"2020-10-01","arxiv_id":"2010.00581","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-social-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2010.00581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.00581"}},"official":null}},{"url":"/paper/pettingzoo-gym-for-multi-agent-reinforcement","slug":"pettingzoo-gym-for-multi-agent-reinforcement","title":"PettingZoo: Gym for Multi-Agent Reinforcement Learning","date":"2020-09-30","arxiv_id":"2009.14471","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pettingzoo-gym-for-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.14471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.14471"}},"official":{"repos":["Farama-Foundation/PettingZoo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-effective-context-for-meta","slug":"towards-effective-context-for-meta","title":"Towards Effective Context for Meta-Reinforcement Learning: an Approach based on Contrastive Learning","date":"2020-09-29","arxiv_id":"2009.13891","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-effective-context-for-meta#ran","syntology_url":"https://syntology.ai/paper/2009.13891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.13891"}},"official":{"repos":["TJU-DRL-LAB/self-supervised-rl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-unknown","slug":"deep-reinforcement-learning-for-unknown","title":"Toward Deep Supervised Anomaly Detection: Reinforcement Learning from Partially Labeled Anomaly Data","date":"2020-09-15","arxiv_id":"2009.06847","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-unknown#ran","syntology_url":"https://syntology.ai/paper/2009.06847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.06847"}},"official":null}},{"url":"/paper/decoupling-representation-learning-from","slug":"decoupling-representation-learning-from","title":"Decoupling Representation Learning from Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.08319","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-representation-learning-from#ran","syntology_url":"https://syntology.ai/paper/2009.08319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.08319"}},"official":{"repos":["astooke/rlpyt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynode-neural-ordinary-differential-equations","slug":"dynode-neural-ordinary-differential-equations","title":"DyNODE: Neural Ordinary Differential Equations for Dynamics Modeling in Continuous Control","date":"2020-09-09","arxiv_id":"2009.04278","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynode-neural-ordinary-differential-equations#ran","syntology_url":"https://syntology.ai/paper/2009.04278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.04278"}},"official":{"repos":["vmartinezalvarez/DyNODE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/confuciux-autonomous-hardware-resource","slug":"confuciux-autonomous-hardware-resource","title":"ConfuciuX: Autonomous Hardware Resource Assignment for DNN Accelerators using Reinforcement Learning","date":"2020-09-04","arxiv_id":"2009.02010","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/confuciux-autonomous-hardware-resource#ran","syntology_url":"https://syntology.ai/paper/2009.02010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.02010"}},"official":{"repos":["maestro-project/confuciux"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-automated-deep-reinforcement","slug":"sample-efficient-automated-deep-reinforcement","title":"Sample-Efficient Automated Deep Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01555","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-automated-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2009.01555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.01555"}},"official":{"repos":["automl/SEARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-off-policy-with-online-planning","slug":"learning-off-policy-with-online-planning","title":"Learning Off-Policy with Online Planning","date":"2020-08-23","arxiv_id":"2008.10066","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-off-policy-with-online-planning#ran","syntology_url":"https://syntology.ai/paper/2008.10066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.10066"}},"official":null}},{"url":"/paper/a-composable-specification-language-for-1","slug":"a-composable-specification-language-for-1","title":"A Composable Specification Language for Reinforcement Learning Tasks","date":"2020-08-21","arxiv_id":"2008.09293","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-composable-specification-language-for-1#ran","syntology_url":"https://syntology.ai/paper/2008.09293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.09293"}},"official":{"repos":["keyshor/spectrl_tool"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-reinforcement-learning-in-constrained","slug":"safe-reinforcement-learning-in-constrained","title":"Safe Reinforcement Learning in Constrained Markov Decision Processes","date":"2020-08-15","arxiv_id":"2008.06626","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-reinforcement-learning-in-constrained#ran","syntology_url":"https://syntology.ai/paper/2008.06626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06626"}},"official":{"repos":["akifumi-wachi-4/safe_near_optimal_mdp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-cross-entropy-method-for","slug":"sample-efficient-cross-entropy-method-for","title":"Sample-efficient Cross-Entropy Method for Real-time Planning","date":"2020-08-14","arxiv_id":"2008.06389","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sample-efficient-cross-entropy-method-for#ran","syntology_url":"https://syntology.ai/paper/2008.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06389"}},"official":{"repos":["martius-lab/iCEM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-meta-reinforcement-learning-with","slug":"offline-meta-reinforcement-learning-with","title":"Offline Meta-Reinforcement Learning with Advantage Weighting","date":"2020-08-13","arxiv_id":"2008.06043","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/offline-meta-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2008.06043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06043"}},"official":{"repos":["eric-mitchell/macaw"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"bb74149cd94096f233d6023b372b3cffad0812c410f2bcf418c19e5977743287","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}