{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/4","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":132,"rows_per_page":100,"rows":[301,400],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/3","next":"/task/reinforcement-learning/papers/5","papers":[{"url":"/paper/deep-reinforcement-learning-for-turbulence","slug":"deep-reinforcement-learning-for-turbulence","title":"Deep Reinforcement Learning for Turbulence Modeling in Large Eddy Simulations","date":"2022-06-21","arxiv_id":"2206.11038","repositories_listed":3,"syntology":null},{"url":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envpool-a-highly-parallel-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.10558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10558"}},"official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/a-unified-approach-to-reinforcement-learning","slug":"a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-approach-to-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05825"}},"official":{"repos":["deepmind/open_spiel"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/mildly-conservative-q-learning-for-offline","slug":"mildly-conservative-q-learning-for-offline","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04745","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":6,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":11,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mildly-conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.04745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04745"}},"official":{"repos":["dmksjfl/mcq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/supported-policy-optimization-for-offline","slug":"supported-policy-optimization-for-offline","title":"Supported Policy Optimization for Offline Reinforcement Learning","date":"2022-02-13","arxiv_id":"2202.06239","repositories_listed":3,"syntology":null},{"url":"/paper/adversarially-trained-actor-critic-for","slug":"adversarially-trained-actor-critic-for","title":"Adversarially Trained Actor Critic for Offline Reinforcement Learning","date":"2022-02-05","arxiv_id":"2202.02446","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarially-trained-actor-critic-for#ran","syntology_url":"https://syntology.ai/paper/2202.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02446"}},"official":{"repos":["microsoft/atac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":4,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-at-the-edge-of#ran","syntology_url":"https://syntology.ai/paper/2108.13264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13264"}},"official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-urban-driving-by-imitating-a","slug":"end-to-end-urban-driving-by-imitating-a","title":"End-to-End Urban Driving by Imitating a Reinforcement Learning Coach","date":"2021-08-18","arxiv_id":"2108.08265","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-urban-driving-by-imitating-a#ran","syntology_url":"https://syntology.ai/paper/2108.08265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.08265"}},"official":{"repos":["zhejz/carla-roach"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/androidenv-a-reinforcement-learning-platform","slug":"androidenv-a-reinforcement-learning-platform","title":"AndroidEnv: A Reinforcement Learning Platform for Android","date":"2021-05-27","arxiv_id":"2105.13231","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/androidenv-a-reinforcement-learning-platform#ran","syntology_url":"https://syntology.ai/paper/2105.13231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13231"}},"official":{"repos":["deepmind/android_env"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/feasible-actor-critic-constrained","slug":"feasible-actor-critic-constrained","title":"Feasible Actor-Critic: Constrained Reinforcement Learning for Ensuring Statewise Safety","date":"2021-05-22","arxiv_id":"2105.10682","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/feasible-actor-critic-constrained#ran","syntology_url":"https://syntology.ai/paper/2105.10682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10682"}},"official":{"repos":["mahaitongdae/Feasible-Actor-Critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/constructions-in-combinatorics-via-neural","slug":"constructions-in-combinatorics-via-neural","title":"Constructions in combinatorics via neural networks","date":"2021-04-29","arxiv_id":"2104.14516","repositories_listed":3,"syntology":null},{"url":"/paper/podracer-architectures-for-scalable","slug":"podracer-architectures-for-scalable","title":"Podracer architectures for scalable Reinforcement Learning","date":"2021-04-13","arxiv_id":"2104.06272","repositories_listed":3,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/podracer-architectures-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2104.06272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06272"}},"official":null}},{"url":"/paper/learning-to-fly-a-gym-environment-with","slug":"learning-to-fly-a-gym-environment-with","title":"Learning to Fly -- a Gym Environment with PyBullet Physics for Reinforcement Learning of Multi-agent Quadcopter Control","date":"2021-03-03","arxiv_id":"2103.02142","repositories_listed":3,"syntology":null},{"url":"/paper/near-real-world-benchmarks-for-offline","slug":"near-real-world-benchmarks-for-offline","title":"NeoRL: A Near Real-World Benchmark for Offline Reinforcement Learning","date":"2021-02-01","arxiv_id":"2102.00714","repositories_listed":3,"syntology":null},{"url":"/paper/learning-fair-policies-in-decentralized","slug":"learning-fair-policies-in-decentralized","title":"Learning Fair Policies in Decentralized Cooperative Multi-Agent Reinforcement Learning","date":"2020-12-17","arxiv_id":"2012.09421","repositories_listed":3,"syntology":null},{"url":"/paper/generalization-in-reinforcement-learning-by","slug":"generalization-in-reinforcement-learning-by","title":"Generalization in Reinforcement Learning by Soft Data Augmentation","date":"2020-11-26","arxiv_id":"2011.13389","repositories_listed":3,"syntology":null},{"url":"/paper/pomo-policy-optimization-with-multiple-optima","slug":"pomo-policy-optimization-with-multiple-optima","title":"POMO: Policy Optimization with Multiple Optima for Reinforcement Learning","date":"2020-10-30","arxiv_id":"2010.16011","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/pomo-policy-optimization-with-multiple-optima#ran","syntology_url":"https://syntology.ai/paper/2010.16011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.16011"}},"official":{"repos":["yd-kwon/POMO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-with-random-delays-1","slug":"reinforcement-learning-with-random-delays-1","title":"Reinforcement Learning with Random Delays","date":"2020-10-06","arxiv_id":"2010.02966","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-random-delays-1#ran","syntology_url":"https://syntology.ai/paper/2010.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02966"}},"official":{"repos":["rmst/rlrd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reward-machines-exploiting-reward-function","slug":"reward-machines-exploiting-reward-function","title":"Reward Machines: Exploiting Reward Function Structure in Reinforcement Learning","date":"2020-10-06","arxiv_id":"2010.03950","repositories_listed":3,"syntology":null},{"url":"/paper/decoupling-representation-learning-from","slug":"decoupling-representation-learning-from","title":"Decoupling Representation Learning from Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.08319","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-representation-learning-from#ran","syntology_url":"https://syntology.ai/paper/2009.08319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.08319"}},"official":{"repos":["astooke/rlpyt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flightmare-a-flexible-quadrotor-simulator","slug":"flightmare-a-flexible-quadrotor-simulator","title":"Flightmare: A Flexible Quadrotor Simulator","date":"2020-09-01","arxiv_id":"2009.00563","repositories_listed":3,"syntology":null},{"url":"/paper/or-gym-a-reinforcement-learning-library-for","slug":"or-gym-a-reinforcement-learning-library-for","title":"OR-Gym: A Reinforcement Learning Library for Operations Research Problems","date":"2020-08-14","arxiv_id":"2008.06319","repositories_listed":3,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/or-gym-a-reinforcement-learning-library-for#ran","syntology_url":"https://syntology.ai/paper/2008.06319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06319"}},"official":{"repos":["hubbs5/or-gym"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/babyai-1-1","slug":"babyai-1-1","title":"BabyAI 1.1","date":"2020-07-24","arxiv_id":"2007.12770","repositories_listed":3,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/babyai-1-1#ran","syntology_url":"https://syntology.ai/paper/2007.12770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12770"}},"official":{"repos":["mila-iqia/babyai"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/implicit-distributional-reinforcement","slug":"implicit-distributional-reinforcement","title":"Implicit Distributional Reinforcement Learning","date":"2020-07-13","arxiv_id":"2007.06159","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/implicit-distributional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2007.06159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06159"}},"official":{"repos":["zhougroup/IDAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-soft-advantage-fitting-imitation","slug":"adversarial-soft-advantage-fitting-imitation","title":"Adversarial Soft Advantage Fitting: Imitation Learning without Policy Optimization","date":"2020-06-23","arxiv_id":"2006.13258","repositories_listed":3,"syntology":null},{"url":"/paper/shared-experience-actor-critic-for-multi","slug":"shared-experience-actor-critic-for-multi","title":"Shared Experience Actor-Critic for Multi-Agent Reinforcement Learning","date":"2020-06-12","arxiv_id":"2006.07169","repositories_listed":3,"syntology":null},{"url":"/paper/offline-reinforcement-learning-tutorial","slug":"offline-reinforcement-learning-tutorial","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","date":"2020-05-04","arxiv_id":"2005.01643","repositories_listed":3,"syntology":null},{"url":"/paper/ultrasound-guided-robotic-navigation-with","slug":"ultrasound-guided-robotic-navigation-with","title":"Ultrasound-Guided Robotic Navigation with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13321","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ultrasound-guided-robotic-navigation-with#ran","syntology_url":"https://syntology.ai/paper/2003.13321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13321"}},"official":{"repos":["hhase/spinal-navigation-rl","hhase/sacrum_data-set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/neuroevolution-of-self-interpretable-agents","slug":"neuroevolution-of-self-interpretable-agents","title":"Neuroevolution of Self-Interpretable Agents","date":"2020-03-18","arxiv_id":"2003.08165","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neuroevolution-of-self-interpretable-agents#ran","syntology_url":"https://syntology.ai/paper/2003.08165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08165"}},"official":null}},{"url":"/paper/deep-multi-agent-reinforcement-learning-for","slug":"deep-multi-agent-reinforcement-learning-for","title":"FACMAC: Factored Multi-Agent Centralised Policy Gradients","date":"2020-03-14","arxiv_id":"2003.06709","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-multi-agent-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2003.06709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06709"}},"official":{"repos":["schroederdewitt/multiagent_mujoco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-reinforcement-learning","slug":"sample-efficient-reinforcement-learning","title":"Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","date":"2020-03-12","arxiv_id":"2003.06066","repositories_listed":3,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sample-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.06066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06066"}},"official":null}},{"url":"/paper/tadam-a-robust-stochastic-gradient-optimizer","slug":"tadam-a-robust-stochastic-gradient-optimizer","title":"TAdam: A Robust Stochastic Gradient Optimizer","date":"2020-02-29","arxiv_id":"2003.00179","repositories_listed":3,"syntology":null},{"url":"/paper/jelly-bean-world-a-testbed-for-never-ending-1","slug":"jelly-bean-world-a-testbed-for-never-ending-1","title":"Jelly Bean World: A Testbed for Never-Ending Learning","date":"2020-02-15","arxiv_id":"2002.06306","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jelly-bean-world-a-testbed-for-never-ending-1#ran","syntology_url":"https://syntology.ai/paper/2002.06306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06306"}},"official":{"repos":["eaplatanios/jelly-bean-world"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-deep-reinforcement-learning-algorithm-using","slug":"a-deep-reinforcement-learning-algorithm-using","title":"A Deep Reinforcement Learning Algorithm Using Dynamic Attention Model for Vehicle Routing Problems","date":"2020-02-09","arxiv_id":"2002.03282","repositories_listed":3,"syntology":null},{"url":"/paper/addressing-value-estimation-errors-in","slug":"addressing-value-estimation-errors-in","title":"Distributional Soft Actor-Critic: Off-Policy Reinforcement Learning for Addressing Value Estimation Errors","date":"2020-01-09","arxiv_id":"2001.02811","repositories_listed":3,"syntology":null},{"url":"/paper/generative-teaching-networks-accelerating-1","slug":"generative-teaching-networks-accelerating-1","title":"Generative Teaching Networks: Accelerating Neural Architecture Search by Learning to Generate Synthetic Training Data","date":"2019-12-17","arxiv_id":"1912.07768","repositories_listed":3,"syntology":null},{"url":"/paper/imitation-learning-via-off-policy-1","slug":"imitation-learning-via-off-policy-1","title":"Imitation Learning via Off-Policy Distribution Matching","date":"2019-12-10","arxiv_id":"1912.05032","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-object-detection-in-large-images","slug":"efficient-object-detection-in-large-images","title":"Efficient Object Detection in Large Images using Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.03966","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-object-detection-in-large-images#ran","syntology_url":"https://syntology.ai/paper/1912.03966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03966"}},"official":{"repos":["uzkent/EfficientObjectDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reinforcement-learning-upside-down-dont","slug":"reinforcement-learning-upside-down-dont","title":"Reinforcement Learning Upside Down: Don't Predict Rewards -- Just Map Them to Actions","date":"2019-12-05","arxiv_id":"1912.02875","repositories_listed":3,"syntology":null},{"url":"/paper/empirical-study-of-off-policy-policy","slug":"empirical-study-of-off-policy-policy","title":"Empirical Study of Off-Policy Policy Evaluation for Reinforcement Learning","date":"2019-11-15","arxiv_id":"1911.06854","repositories_listed":3,"syntology":null},{"url":"/paper/real-time-reinforcement-learning","slug":"real-time-reinforcement-learning","title":"Real-Time Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04448","repositories_listed":3,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.04448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04448"}},"official":{"repos":["rmst/rtrl","elementai/avenue"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-model-based-reinforcement-learning-with","slug":"a-model-based-reinforcement-learning-with","title":"Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-11-10","arxiv_id":"1911.03845","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-model-based-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1911.03845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.03845"}},"official":{"repos":["JianGuanTHU/IRecGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reinforcement-learn-for-neural","slug":"learning-to-reinforcement-learn-for-neural","title":"Learning to reinforcement learn for Neural Architecture Search","date":"2019-11-09","arxiv_id":"1911.03769","repositories_listed":3,"syntology":null},{"url":"/paper/a-divergence-minimization-perspective-on","slug":"a-divergence-minimization-perspective-on","title":"A Divergence Minimization Perspective on Imitation Learning Methods","date":"2019-11-06","arxiv_id":"1911.02256","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-divergence-minimization-perspective-on#ran","syntology_url":"https://syntology.ai/paper/1911.02256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02256"}},"official":{"repos":["KamyarGh/rl_swiss"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bananas-bayesian-optimization-with-neural","slug":"bananas-bayesian-optimization-with-neural","title":"BANANAS: Bayesian Optimization with Neural Architectures for Neural Architecture Search","date":"2019-10-25","arxiv_id":"1910.11858","repositories_listed":3,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bananas-bayesian-optimization-with-neural#ran","syntology_url":"https://syntology.ai/paper/1910.11858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11858"}},"official":{"repos":["naszilla/bananas","naszilla/naszilla"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/collision-avoidance-in-pedestrian-rich","slug":"collision-avoidance-in-pedestrian-rich","title":"Collision Avoidance in Pedestrian-Rich Environments with Deep Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.11689","repositories_listed":3,"syntology":null},{"url":"/paper/torchbeast-a-pytorch-platform-for-distributed","slug":"torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/torchbeast-a-pytorch-platform-for-distributed#ran","syntology_url":"https://syntology.ai/paper/1910.03552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.03552"}},"official":{"repos":["heiner/scalable_agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/policies-modulating-trajectory-generators","slug":"policies-modulating-trajectory-generators","title":"Policies Modulating Trajectory Generators","date":"2019-10-07","arxiv_id":"1910.02812","repositories_listed":3,"syntology":null},{"url":"/paper/generalized-inner-loop-meta-learning","slug":"generalized-inner-loop-meta-learning","title":"Generalized Inner Loop Meta-Learning","date":"2019-10-03","arxiv_id":"1910.01727","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generalized-inner-loop-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1910.01727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01727"}},"official":{"repos":["learnables/learn2learn","facebookresearch/higher"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reducing-overestimation-bias-in-multi-agent","slug":"reducing-overestimation-bias-in-multi-agent","title":"Reducing Overestimation Bias in Multi-Agent Domains Using Double Centralized Critics","date":"2019-10-03","arxiv_id":"1910.01465","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reducing-overestimation-bias-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1910.01465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01465"}},"official":{"repos":["JohannesAck/MATD3implementation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/scalable-global-optimization-via-local","slug":"scalable-global-optimization-via-local","title":"Scalable Global Optimization via Local Bayesian Optimization","date":"2019-10-03","arxiv_id":"1910.01739","repositories_listed":3,"syntology":null},{"url":"/paper/c-3po-cyclic-three-phase-optimization-for","slug":"c-3po-cyclic-three-phase-optimization-for","title":"C-3PO: Cyclic-Three-Phase Optimization for Human-Robot Motion Retargeting based on Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11303","repositories_listed":3,"syntology":null},{"url":"/paper/emergent-tool-use-from-multi-agent","slug":"emergent-tool-use-from-multi-agent","title":"Emergent Tool Use From Multi-Agent Autocurricula","date":"2019-09-17","arxiv_id":"1909.07528","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergent-tool-use-from-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1909.07528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07528"}},"official":{"repos":["openai/multi-agent-emergence-environments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/approximating-two-value-functions-instead-of","slug":"approximating-two-value-functions-instead-of","title":"Approximating two value functions instead of one: towards characterizing a new family of Deep Reinforcement Learning algorithms","date":"2019-09-01","arxiv_id":"1909.01779","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-play-the-chess-variant-crazyhouse","slug":"learning-to-play-the-chess-variant-crazyhouse","title":"Learning to play the Chess Variant Crazyhouse above World Champion Level with Deep Neural Networks and Human Data","date":"2019-08-19","arxiv_id":"1908.06660","repositories_listed":3,"syntology":null},{"url":"/paper/behaviour-suite-for-reinforcement-learning","slug":"behaviour-suite-for-reinforcement-learning","title":"Behaviour Suite for Reinforcement Learning","date":"2019-08-09","arxiv_id":"1908.03568","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behaviour-suite-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1908.03568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.03568"}},"official":{"repos":["deepmind/bsuite"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/dynamics-aware-unsupervised-discovery-of","slug":"dynamics-aware-unsupervised-discovery-of","title":"Dynamics-Aware Unsupervised Discovery of Skills","date":"2019-07-02","arxiv_id":"1907.01657","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamics-aware-unsupervised-discovery-of#ran","syntology_url":"https://syntology.ai/paper/1907.01657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01657"}},"official":{"repos":["google-research/dads"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/placeto-learning-generalizable-device","slug":"placeto-learning-generalizable-device","title":"Placeto: Learning Generalizable Device Placement Algorithms for Distributed Machine Learning","date":"2019-06-20","arxiv_id":"1906.08879","repositories_listed":3,"syntology":null},{"url":"/paper/nas-fcos-fast-neural-architecture-search-for","slug":"nas-fcos-fast-neural-architecture-search-for","title":"NAS-FCOS: Fast Neural Architecture Search for Object Detection","date":"2019-06-11","arxiv_id":"1906.04423","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nas-fcos-fast-neural-architecture-search-for#ran","syntology_url":"https://syntology.ai/paper/1906.04423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04423"}},"official":null}},{"url":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-soft-actor-critic-emphasizing-recent#ran","syntology_url":"https://syntology.ai/paper/1906.04009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04009"}},"official":null}},{"url":"/paper/machine-learning-and-system-identification","slug":"machine-learning-and-system-identification","title":"Machine Learning and System Identification for Estimation in Physical Systems","date":"2019-06-05","arxiv_id":"1906.02003","repositories_listed":3,"syntology":null},{"url":"/paper/190600949","slug":"190600949","title":"Stabilizing Off-Policy Q-Learning via Bootstrapping Error Reduction","date":"2019-06-03","arxiv_id":"1906.00949","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/190600949#ran","syntology_url":"https://syntology.ai/paper/1906.00949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.00949"}},"official":null}},{"url":"/paper/reinforcement-learning-for-slate-based","slug":"reinforcement-learning-for-slate-based","title":"Reinforcement Learning for Slate-based Recommender Systems: A Tractable Decomposition and Practical Methodology","date":"2019-05-29","arxiv_id":"1905.12767","repositories_listed":3,"syntology":null},{"url":"/paper/maximum-entropy-regularized-multi-goal","slug":"maximum-entropy-regularized-multi-goal","title":"Maximum Entropy-Regularized Multi-Goal Reinforcement Learning","date":"2019-05-21","arxiv_id":"1905.08786","repositories_listed":3,"syntology":null},{"url":"/paper/recurrent-experience-replay-in-distributed","slug":"recurrent-experience-replay-in-distributed","title":"Recurrent Experience Replay in Distributed Reinforcement Learning","date":"2019-05-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/end-to-end-robotic-reinforcement-learning","slug":"end-to-end-robotic-reinforcement-learning","title":"End-to-End Robotic Reinforcement Learning without Reward Engineering","date":"2019-04-16","arxiv_id":"1904.07854","repositories_listed":3,"syntology":null},{"url":"/paper/extrapolating-beyond-suboptimal","slug":"extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extrapolating-beyond-suboptimal#ran","syntology_url":"https://syntology.ai/paper/1904.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06387"}},"official":{"repos":["hiwonjoon/ICML2019-TREX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/holist-an-environment-for-machine-learning-of","slug":"holist-an-environment-for-machine-learning-of","title":"HOList: An Environment for Machine Learning of Higher-Order Theorem Proving","date":"2019-04-05","arxiv_id":"1904.03241","repositories_listed":3,"syntology":null},{"url":"/paper/improved-robustness-of-reinforcement-learning","slug":"improved-robustness-of-reinforcement-learning","title":"Improved robustness of reinforcement learning policies upon conversion to spiking neuronal network platforms applied to ATARI games","date":"2019-03-26","arxiv_id":"1903.11012","repositories_listed":3,"syntology":null},{"url":"/paper/minatar-an-atari-inspired-testbed-for-more","slug":"minatar-an-atari-inspired-testbed-for-more","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","date":"2019-03-07","arxiv_id":"1903.03176","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/minatar-an-atari-inspired-testbed-for-more#ran","syntology_url":"https://syntology.ai/paper/1903.03176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03176"}},"official":{"repos":["kenjyoung/MinAtar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/rethinking-action-spaces-for-reinforcement","slug":"rethinking-action-spaces-for-reinforcement","title":"Rethinking Action Spaces for Reinforcement Learning in End-to-end Dialog Agents with Latent Variable Models","date":"2019-02-23","arxiv_id":"1902.08858","repositories_listed":3,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rethinking-action-spaces-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1902.08858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.08858"}},"official":{"repos":["snakeztc/NeuralDialog-LaRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-critics-assignment-for-multi","slug":"hierarchical-critics-assignment-for-multi","title":"Reinforcement Learning from Hierarchical Critics","date":"2019-02-08","arxiv_id":"1902.03079","repositories_listed":3,"syntology":null},{"url":"/paper/pipps-flexible-model-based-policy-search","slug":"pipps-flexible-model-based-policy-search","title":"PIPPS: Flexible Model-Based Policy Search Robust to the Curse of Chaos","date":"2019-02-04","arxiv_id":"1902.01240","repositories_listed":3,"syntology":null},{"url":"/paper/go-explore-a-new-approach-for-hard","slug":"go-explore-a-new-approach-for-hard","title":"Go-Explore: a New Approach for Hard-Exploration Problems","date":"2019-01-30","arxiv_id":"1901.10995","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/go-explore-a-new-approach-for-hard#ran","syntology_url":"https://syntology.ai/paper/1901.10995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.10995"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-imbalanced#ran","syntology_url":"https://syntology.ai/paper/1901.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01379"}},"official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/toybox-better-atari-environments-for-testing","slug":"toybox-better-atari-environments-for-testing","title":"ToyBox: Better Atari Environments for Testing Reinforcement Learning Agents","date":"2018-12-06","arxiv_id":"1812.02850","repositories_listed":3,"syntology":null},{"url":"/paper/scalable-agent-alignment-via-reward-modeling","slug":"scalable-agent-alignment-via-reward-modeling","title":"Scalable agent alignment via reward modeling: a research direction","date":"2018-11-19","arxiv_id":"1811.07871","repositories_listed":3,"syntology":null},{"url":"/paper/quota-the-quantile-option-architecture-for","slug":"quota-the-quantile-option-architecture-for","title":"QUOTA: The Quantile Option Architecture for Reinforcement Learning","date":"2018-11-05","arxiv_id":"1811.02073","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quota-the-quantile-option-architecture-for#ran","syntology_url":"https://syntology.ai/paper/1811.02073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.02073"}},"official":{"repos":["ShangtongZhang/DeepRL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-to-learn-without-forgetting-by","slug":"learning-to-learn-without-forgetting-by","title":"Learning to Learn without Forgetting by Maximizing Transfer and Minimizing Interference","date":"2018-10-29","arxiv_id":"1810.11910","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-learn-without-forgetting-by#ran","syntology_url":"https://syntology.ai/paper/1810.11910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11910"}},"official":{"repos":["mattriemer/mer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/social-influence-as-intrinsic-motivation-for","slug":"social-influence-as-intrinsic-motivation-for","title":"Social Influence as Intrinsic Motivation for Multi-Agent Deep Reinforcement Learning","date":"2018-10-19","arxiv_id":"1810.08647","repositories_listed":3,"syntology":null},{"url":"/paper/actor-attention-critic-for-multi-agent","slug":"actor-attention-critic-for-multi-agent","title":"Actor-Attention-Critic for Multi-Agent Reinforcement Learning","date":"2018-10-05","arxiv_id":"1810.02912","repositories_listed":3,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-attention-critic-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1810.02912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02912"}},"official":{"repos":["shariqiqbal2810/MAAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-quality-value-dqv-learning","slug":"deep-quality-value-dqv-learning","title":"Deep Quality-Value (DQV) Learning","date":"2018-09-30","arxiv_id":"1810.00368","repositories_listed":3,"syntology":null},{"url":"/paper/dynamic-weights-in-multi-objective-deep","slug":"dynamic-weights-in-multi-objective-deep","title":"Dynamic Weights in Multi-Objective Deep Reinforcement Learning","date":"2018-09-20","arxiv_id":"1809.07803","repositories_listed":3,"syntology":null},{"url":"/paper/jointly-learning-to-see-ask-and-guesswhat","slug":"jointly-learning-to-see-ask-and-guesswhat","title":"Beyond task success: A closer look at jointly learning to see, ask, and GuessWhat","date":"2018-09-10","arxiv_id":"1809.03408","repositories_listed":3,"syntology":null},{"url":"/paper/discriminator-actor-critic-addressing-sample","slug":"discriminator-actor-critic-addressing-sample","title":"Discriminator-Actor-Critic: Addressing Sample Inefficiency and Reward Bias in Adversarial Imitation Learning","date":"2018-09-09","arxiv_id":"1809.02925","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":4,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 4 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discriminator-actor-critic-addressing-sample#ran","syntology_url":"https://syntology.ai/paper/1809.02925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.02925"}},"official":null}},{"url":"/paper/multi-hop-knowledge-graph-reasoning-with","slug":"multi-hop-knowledge-graph-reasoning-with","title":"Multi-Hop Knowledge Graph Reasoning with Reward Shaping","date":"2018-08-31","arxiv_id":"1808.10568","repositories_listed":3,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-hop-knowledge-graph-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/1808.10568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.10568"}},"official":{"repos":["salesforce/MultiHopKG"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/decoupling-strategy-and-generation-in","slug":"decoupling-strategy-and-generation-in","title":"Decoupling Strategy and Generation in Negotiation Dialogues","date":"2018-08-29","arxiv_id":"1808.09637","repositories_listed":3,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/decoupling-strategy-and-generation-in#ran","syntology_url":"https://syntology.ai/paper/1808.09637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09637"}},"official":{"repos":["worksheets.codalab.org/worksheets/0x453913e76b65495d8b9730d41c7e0a0c"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bipedal-walking-robot-using-deep","slug":"bipedal-walking-robot-using-deep","title":"Bipedal Walking Robot using Deep Deterministic Policy Gradient","date":"2018-07-16","arxiv_id":"1807.05924","repositories_listed":3,"syntology":null},{"url":"/paper/evolving-multimodal-robot-behavior-via-many","slug":"evolving-multimodal-robot-behavior-via-many","title":"Evolving Multimodal Robot Behavior via Many Stepping Stones with the Combinatorial Multi-Objective Evolutionary Algorithm","date":"2018-07-09","arxiv_id":"1807.03392","repositories_listed":3,"syntology":null},{"url":"/paper/scheduled-policy-optimization-for-natural","slug":"scheduled-policy-optimization-for-natural","title":"Scheduled Policy Optimization for Natural Language Communication with Intelligent Agents","date":"2018-06-16","arxiv_id":"1806.06187","repositories_listed":3,"syntology":null},{"url":"/paper/surprising-negative-results-for-generative","slug":"surprising-negative-results-for-generative","title":"Surprising Negative Results for Generative Adversarial Tree Search","date":"2018-06-15","arxiv_id":"1806.05780","repositories_listed":3,"syntology":null},{"url":"/paper/maximum-a-posteriori-policy-optimisation","slug":"maximum-a-posteriori-policy-optimisation","title":"Maximum a Posteriori Policy Optimisation","date":"2018-06-14","arxiv_id":"1806.06920","repositories_listed":3,"syntology":null},{"url":"/paper/fast-and-scalable-bayesian-deep-learning-by","slug":"fast-and-scalable-bayesian-deep-learning-by","title":"Fast and Scalable Bayesian Deep Learning by Weight-Perturbation in Adam","date":"2018-06-13","arxiv_id":"1806.04854","repositories_listed":3,"syntology":null},{"url":"/paper/path-level-network-transformation-for","slug":"path-level-network-transformation-for","title":"Path-Level Network Transformation for Efficient Architecture Search","date":"2018-06-07","arxiv_id":"1806.02639","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/path-level-network-transformation-for#ran","syntology_url":"https://syntology.ai/paper/1806.02639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02639"}},"official":{"repos":["han-cai/PathLevel-EAS"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/transfer-learning-for-related-reinforcement","slug":"transfer-learning-for-related-reinforcement","title":"Transfer Learning for Related Reinforcement Learning Tasks via Image-to-Image Translation","date":"2018-05-31","arxiv_id":"1806.07377","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sequence-to","slug":"deep-reinforcement-learning-for-sequence-to","title":"Deep Reinforcement Learning For Sequence to Sequence Models","date":"2018-05-24","arxiv_id":"1805.09461","repositories_listed":3,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sequence-to#ran","syntology_url":"https://syntology.ai/paper/1805.09461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09461"}},"official":{"repos":["yaserkl/RLSeq2Seq"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/toward-diverse-text-generation-with-inverse","slug":"toward-diverse-text-generation-with-inverse","title":"Toward Diverse Text Generation with Inverse Reinforcement Learning","date":"2018-04-30","arxiv_id":"1804.11258","repositories_listed":3,"syntology":null},{"url":"/paper/gotta-learn-fast-a-new-benchmark-for","slug":"gotta-learn-fast-a-new-benchmark-for","title":"Gotta Learn Fast: A New Benchmark for Generalization in RL","date":"2018-04-10","arxiv_id":"1804.03720","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-traffic-light","slug":"deep-reinforcement-learning-for-traffic-light","title":"Deep Reinforcement Learning for Traffic Light Control in Vehicular Networks","date":"2018-03-29","arxiv_id":"1803.11115","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-interactive-annotation-of","slug":"efficient-interactive-annotation-of","title":"Efficient Interactive Annotation of Segmentation Datasets with Polygon-RNN++","date":"2018-03-26","arxiv_id":"1803.09693","repositories_listed":3,"syntology":null}],"record_sha256":"b302235901cc28d430b7f0def703b3b0e2f581449bbc4bf22ddecf6a10863540","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}