{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/3","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":132,"rows_per_page":100,"rows":[201,300],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/2","next":"/task/reinforcement-learning/papers/4","papers":[{"url":"/paper/regularized-evolution-for-image-classifier","slug":"regularized-evolution-for-image-classifier","title":"Regularized Evolution for Image Classifier Architecture Search","date":"2018-02-05","arxiv_id":"1802.01548","repositories_listed":5,"syntology":null},{"url":"/paper/building-generalizable-agents-with-a","slug":"building-generalizable-agents-with-a","title":"Building Generalizable Agents with a Realistic and Rich 3D Environment","date":"2018-01-07","arxiv_id":"1801.02209","repositories_listed":5,"syntology":null},{"url":"/paper/action-branching-architectures-for-deep","slug":"action-branching-architectures-for-deep","title":"Action Branching Architectures for Deep Reinforcement Learning","date":"2017-11-24","arxiv_id":"1711.08946","repositories_listed":5,"syntology":null},{"url":"/paper/learning-to-generalize-meta-learning-for","slug":"learning-to-generalize-meta-learning-for","title":"Learning to Generalize: Meta-Learning for Domain Generalization","date":"2017-10-10","arxiv_id":"1710.03463","repositories_listed":5,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-generalize-meta-learning-for#ran","syntology_url":"https://syntology.ai/paper/1710.03463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.03463"}},"official":null}},{"url":"/paper/deep-learning-based-numerical-methods-for","slug":"deep-learning-based-numerical-methods-for","title":"Deep learning-based numerical methods for high-dimensional parabolic partial differential equations and backward stochastic differential equations","date":"2017-06-15","arxiv_id":"1706.04702","repositories_listed":5,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-learning-based-numerical-methods-for#ran","syntology_url":"https://syntology.ai/paper/1706.04702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1706.04702"}},"official":null}},{"url":"/paper/concrete-dropout","slug":"concrete-dropout","title":"Concrete Dropout","date":"2017-05-22","arxiv_id":"1705.07832","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/concrete-dropout#ran","syntology_url":"https://syntology.ai/paper/1705.07832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1705.07832"}},"official":null}},{"url":"/paper/efficient-parallel-methods-for-deep","slug":"efficient-parallel-methods-for-deep","title":"Efficient Parallel Methods for Deep Reinforcement Learning","date":"2017-05-13","arxiv_id":"1705.04862","repositories_listed":5,"syntology":null},{"url":"/paper/stabilising-experience-replay-for-deep-multi","slug":"stabilising-experience-replay-for-deep-multi","title":"Stabilising Experience Replay for Deep Multi-Agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08887","repositories_listed":5,"syntology":null},{"url":"/paper/designing-neural-network-architectures-using","slug":"designing-neural-network-architectures-using","title":"Designing Neural Network Architectures using Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.02167","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/designing-neural-network-architectures-using#ran","syntology_url":"https://syntology.ai/paper/1611.02167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.02167"}},"official":null}},{"url":"/paper/deep-recurrent-q-learning-for-partially","slug":"deep-recurrent-q-learning-for-partially","title":"Deep Recurrent Q-Learning for Partially Observable MDPs","date":"2015-07-23","arxiv_id":"1507.06527","repositories_listed":5,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/deep-recurrent-q-learning-for-partially#ran","syntology_url":"https://syntology.ai/paper/1507.06527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1507.06527"}},"official":null}},{"url":"/paper/multiple-object-recognition-with-visual","slug":"multiple-object-recognition-with-visual","title":"Multiple Object Recognition with Visual Attention","date":"2014-12-24","arxiv_id":"1412.7755","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-object-recognition-with-visual#ran","syntology_url":"https://syntology.ai/paper/1412.7755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1412.7755"}},"official":null}},{"url":"/paper/hierarchical-reinforcement-learning-with-the","slug":"hierarchical-reinforcement-learning-with-the","title":"Hierarchical Reinforcement Learning with the MAXQ Value Function Decomposition","date":"1999-05-21","arxiv_id":"cs/9905014","repositories_listed":5,"syntology":null},{"url":"/paper/logic-rl-unleashing-llm-reasoning-with-rule","slug":"logic-rl-unleashing-llm-reasoning-with-rule","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","date":"2025-02-20","arxiv_id":"2502.14768","repositories_listed":4,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/logic-rl-unleashing-llm-reasoning-with-rule#ran","syntology_url":"https://syntology.ai/paper/2502.14768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14768"}},"official":{"repos":["Unakar/Logic-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/openrlhf-an-easy-to-use-scalable-and-high","slug":"openrlhf-an-easy-to-use-scalable-and-high","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","date":"2024-05-20","arxiv_id":"2405.11143","repositories_listed":4,"syntology":null},{"url":"/paper/dynamic-datasets-and-market-environments-for","slug":"dynamic-datasets-and-market-environments-for","title":"Dynamic Datasets and Market Environments for Financial Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.13174","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-datasets-and-market-environments-for#ran","syntology_url":"https://syntology.ai/paper/2304.13174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13174"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-in-sample-softmax-for-offline","slug":"the-in-sample-softmax-for-offline","title":"The In-Sample Softmax for Offline Reinforcement Learning","date":"2023-02-28","arxiv_id":"2302.14372","repositories_listed":4,"syntology":null},{"url":"/paper/lifelong-reinforcement-learning-with","slug":"lifelong-reinforcement-learning-with","title":"Lifelong Reinforcement Learning with Modulating Masks","date":"2022-12-21","arxiv_id":"2212.11110","repositories_listed":4,"syntology":null},{"url":"/paper/finrl-meta-market-environments-and-benchmarks","slug":"finrl-meta-market-environments-and-benchmarks","title":"FinRL-Meta: Market Environments and Benchmarks for Data-Driven Financial Reinforcement Learning","date":"2022-11-06","arxiv_id":"2211.03107","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finrl-meta-market-environments-and-benchmarks#ran","syntology_url":"https://syntology.ai/paper/2211.03107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03107"}},"official":{"repos":["AI4Finance-Foundation/FinRL","ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decom-decomposed-policy-for-constrained","slug":"decom-decomposed-policy-for-constrained","title":"DeCOM: Decomposed Policy for Constrained Cooperative Multi-Agent Reinforcement Learning","date":"2021-11-10","arxiv_id":"2111.05670","repositories_listed":4,"syntology":null},{"url":"/paper/multi-agent-constrained-policy-optimisation","slug":"multi-agent-constrained-policy-optimisation","title":"Multi-Agent Constrained Policy Optimisation","date":"2021-10-06","arxiv_id":"2110.02793","repositories_listed":4,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-agent-constrained-policy-optimisation#ran","syntology_url":"https://syntology.ai/paper/2110.02793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02793"}},"official":{"repos":["chauncygu/multi-agent-constrained-policy-optimisation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/safe-control-gym-a-unified-benchmark-suite","slug":"safe-control-gym-a-unified-benchmark-suite","title":"safe-control-gym: a Unified Benchmark Suite for Safe Learning-based Control and Reinforcement Learning in Robotics","date":"2021-09-13","arxiv_id":"2109.06325","repositories_listed":4,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safe-control-gym-a-unified-benchmark-suite#ran","syntology_url":"https://syntology.ai/paper/2109.06325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.06325"}},"official":{"repos":["utiasDSL/safe-control-gym"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/efficient-active-search-for-combinatorial","slug":"efficient-active-search-for-combinatorial","title":"Efficient Active Search for Combinatorial Optimization Problems","date":"2021-06-09","arxiv_id":"2106.05126","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/efficient-active-search-for-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2106.05126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.05126"}},"official":{"repos":["ahottung/EAS"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/objective-robustness-in-deep-reinforcement","slug":"objective-robustness-in-deep-reinforcement","title":"Goal Misgeneralization in Deep Reinforcement Learning","date":"2021-05-28","arxiv_id":"2105.14111","repositories_listed":4,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/objective-robustness-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2105.14111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14111"}},"official":{"repos":["JacobPfau/procgenAISC","jbkjr/train-procgen-pytorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-reinforcement-learning-environment-for-job","slug":"a-reinforcement-learning-environment-for-job","title":"A Reinforcement Learning Environment For Job-Shop Scheduling","date":"2021-04-08","arxiv_id":"2104.03760","repositories_listed":4,"syntology":null},{"url":"/paper/recurrent-rational-networks","slug":"recurrent-rational-networks","title":"Adaptive Rational Activations to Boost Deep Reinforcement Learning","date":"2021-02-18","arxiv_id":"2102.09407","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-rational-networks#ran","syntology_url":"https://syntology.ai/paper/2102.09407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.09407"}},"official":{"repos":["ml-research/rational_activations","ml-research/rational_rl","ml-research/rational_sl","k4ntz/activation-functions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/d2rl-deep-dense-architectures-in-1","slug":"d2rl-deep-dense-architectures-in-1","title":"D2RL: Deep Dense Architectures in Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09163","repositories_listed":4,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/d2rl-deep-dense-architectures-in-1#ran","syntology_url":"https://syntology.ai/paper/2010.09163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09163"}},"official":{"repos":["pairlab/d2rl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["found_in_text","listed","official"]}}},{"url":"/paper/htmrl-biologically-plausible-reinforcement","slug":"htmrl-biologically-plausible-reinforcement","title":"HTMRL: Biologically Plausible Reinforcement Learning with Hierarchical Temporal Memory","date":"2020-09-18","arxiv_id":"2009.08880","repositories_listed":4,"syntology":null},{"url":"/paper/meta-learning-through-hebbian-plasticity-in","slug":"meta-learning-through-hebbian-plasticity-in","title":"Meta-Learning through Hebbian Plasticity in Random Networks","date":"2020-07-06","arxiv_id":"2007.02686","repositories_listed":4,"syntology":null},{"url":"/paper/sample-factory-egocentric-3d-control-from","slug":"sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-factory-egocentric-3d-control-from#ran","syntology_url":"https://syntology.ai/paper/2006.11751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11751"}},"official":{"repos":["alex-petrenko/sample-factory"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deployment-efficient-reinforcement-learning","slug":"deployment-efficient-reinforcement-learning","title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","date":"2020-06-05","arxiv_id":"2006.03647","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deployment-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03647"}},"official":{"repos":["matsuolab/BREMEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-state-dependent-exploration-for#ran","syntology_url":"https://syntology.ai/paper/2005.05719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05719"}},"official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/planning-to-explore-via-self-supervised-world","slug":"planning-to-explore-via-self-supervised-world","title":"Planning to Explore via Self-Supervised World Models","date":"2020-05-12","arxiv_id":"2005.05960","repositories_listed":4,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/planning-to-explore-via-self-supervised-world#ran","syntology_url":"https://syntology.ai/paper/2005.05960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05960"}},"official":{"repos":["ramanans1/plan2explore"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/task-oriented-dialogue-system-for-automatic-1","slug":"task-oriented-dialogue-system-for-automatic-1","title":"Hierarchical Reinforcement Learning for Automatic Disease Diagnosis","date":"2020-04-29","arxiv_id":"2004.14254","repositories_listed":4,"syntology":null},{"url":"/paper/image-augmentation-is-all-you-need","slug":"image-augmentation-is-all-you-need","title":"Image Augmentation Is All You Need: Regularizing Deep Reinforcement Learning from Pixels","date":"2020-04-28","arxiv_id":"2004.13649","repositories_listed":4,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-augmentation-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2004.13649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.13649"}},"official":{"repos":["denisyarats/drq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","repositories_listed":4,"syntology":null},{"url":"/paper/discor-corrective-feedback-in-reinforcement","slug":"discor-corrective-feedback-in-reinforcement","title":"DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","date":"2020-03-16","arxiv_id":"2003.07305","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/discor-corrective-feedback-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.07305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07305"}},"official":null}},{"url":"/paper/ride-rewarding-impact-driven-exploration-for-1","slug":"ride-rewarding-impact-driven-exploration-for-1","title":"RIDE: Rewarding Impact-Driven Exploration for Procedurally-Generated Environments","date":"2020-02-27","arxiv_id":"2002.12292","repositories_listed":4,"syntology":null},{"url":"/paper/accessing-higher-level-representations-in","slug":"accessing-higher-level-representations-in","title":"Addressing Some Limitations of Transformers with Feedback Memory","date":"2020-02-21","arxiv_id":"2002.09402","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-higher-level-representations-in#ran","syntology_url":"https://syntology.ai/paper/2002.09402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.09402"}},"official":{"repos":["facebookresearch/transformer-sequential"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/interpretable-end-to-end-urban-autonomous#ran","syntology_url":"https://syntology.ai/paper/2001.08726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08726"}},"official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/191202288","slug":"191202288","title":"Simplified Action Decoder for Deep Multi-Agent Reinforcement Learning","date":"2019-12-04","arxiv_id":"1912.02288","repositories_listed":4,"syntology":null},{"url":"/paper/comparing-observation-and-action","slug":"comparing-observation-and-action","title":"Comparing Observation and Action Representations for Deep Reinforcement Learning in $μ$RTS","date":"2019-10-26","arxiv_id":"1910.12134","repositories_listed":4,"syntology":null},{"url":"/paper/maven-multi-agent-variational-exploration","slug":"maven-multi-agent-variational-exploration","title":"MAVEN: Multi-Agent Variational Exploration","date":"2019-10-16","arxiv_id":"1910.07483","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/maven-multi-agent-variational-exploration#ran","syntology_url":"https://syntology.ai/paper/1910.07483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07483"}},"official":{"repos":["AnujMahajanOxf/MAVEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/assistive-gym-a-physics-simulation-framework","slug":"assistive-gym-a-physics-simulation-framework","title":"Assistive Gym: A Physics Simulation Framework for Assistive Robotics","date":"2019-10-10","arxiv_id":"1910.04700","repositories_listed":4,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assistive-gym-a-physics-simulation-framework#ran","syntology_url":"https://syntology.ai/paper/1910.04700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04700"}},"official":{"repos":["Healthcare-Robotics/assistive-gym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-sample-efficiency-in-model-free-1","slug":"improving-sample-efficiency-in-model-free-1","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","date":"2019-10-02","arxiv_id":"1910.01741","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-sample-efficiency-in-model-free-1#ran","syntology_url":"https://syntology.ai/paper/1910.01741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01741"}},"official":{"repos":["denisyarats/pytorch_sac_ae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/sme-net-sparse-motion-estimation-for","slug":"sme-net-sparse-motion-estimation-for","title":"SME-Net: Sparse Motion Estimation for Parametric Video Prediction Through Reinforcement Learning","date":"2019-10-01","arxiv_id":null,"repositories_listed":4,"syntology":null},{"url":"/paper/a-generalized-algorithm-for-multi-objective","slug":"a-generalized-algorithm-for-multi-objective","title":"A Generalized Algorithm for Multi-Objective Reinforcement Learning and Policy Adaptation","date":"2019-08-21","arxiv_id":"1908.08342","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-generalized-algorithm-for-multi-objective#ran","syntology_url":"https://syntology.ai/paper/1908.08342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.08342"}},"official":{"repos":["RunzheYang/MORL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/collaborative-multi-agent-dialogue-model","slug":"collaborative-multi-agent-dialogue-model","title":"Collaborative Multi-Agent Dialogue Model Training Via Reinforcement Learning","date":"2019-07-11","arxiv_id":"1907.05507","repositories_listed":4,"syntology":null},{"url":"/paper/learning-transferable-cooperative-behavior-in","slug":"learning-transferable-cooperative-behavior-in","title":"Learning Transferable Cooperative Behavior in Multi-Agent Teams","date":"2019-06-04","arxiv_id":"1906.01202","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-transferable-cooperative-behavior-in#ran","syntology_url":"https://syntology.ai/paper/1906.01202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01202"}},"official":{"repos":["sumitsk/matrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qtran-learning-to-factorize-with","slug":"qtran-learning-to-factorize-with","title":"QTRAN: Learning to Factorize with Transformation for Cooperative Multi-Agent Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05408","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/qtran-learning-to-factorize-with#ran","syntology_url":"https://syntology.ai/paper/1905.05408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05408"}},"official":{"repos":["Sonkyunghwan/QTRAN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/colight-learning-network-level-cooperation","slug":"colight-learning-network-level-cooperation","title":"CoLight: Learning Network-level Cooperation for Traffic Signal Control","date":"2019-05-11","arxiv_id":"1905.05717","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/colight-learning-network-level-cooperation#ran","syntology_url":"https://syntology.ai/paper/1905.05717","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05717"}},"official":{"repos":["wingsweihua/colight"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-pass-q-networks-for-deep-reinforcement","slug":"multi-pass-q-networks-for-deep-reinforcement","title":"Multi-Pass Q-Networks for Deep Reinforcement Learning with Parameterised Action Spaces","date":"2019-05-10","arxiv_id":"1905.04388","repositories_listed":4,"syntology":null},{"url":"/paper/deep-neuroevolution-of-recurrent-and-discrete","slug":"deep-neuroevolution-of-recurrent-and-discrete","title":"Deep Neuroevolution of Recurrent and Discrete World Models","date":"2019-04-28","arxiv_id":"1906.08857","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-neuroevolution-of-recurrent-and-discrete#ran","syntology_url":"https://syntology.ai/paper/1906.08857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.08857"}},"official":{"repos":["sebastianrisi/ga-world-models"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/transferring-knowledge-across-learning","slug":"transferring-knowledge-across-learning","title":"Transferring Knowledge across Learning Processes","date":"2018-12-03","arxiv_id":"1812.01054","repositories_listed":4,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/transferring-knowledge-across-learning#ran","syntology_url":"https://syntology.ai/paper/1812.01054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.01054"}},"official":{"repos":["amzn/xfer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fast-neural-architecture-search-of-compact","slug":"fast-neural-architecture-search-of-compact","title":"Fast Neural Architecture Search of Compact Semantic Segmentation Models via Auxiliary Cells","date":"2018-10-25","arxiv_id":"1810.10804","repositories_listed":4,"syntology":null},{"url":"/paper/graph-convolutional-reinforcement-learning","slug":"graph-convolutional-reinforcement-learning","title":"Graph Convolutional Reinforcement Learning","date":"2018-10-22","arxiv_id":"1810.09202","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-convolutional-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.09202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09202"}},"official":{"repos":["PKU-AI-Edge/DGN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/lift-reinforcement-learning-in-computer","slug":"lift-reinforcement-learning-in-computer","title":"LIFT: Reinforcement Learning in Computer Systems by Learning From Demonstrations","date":"2018-08-23","arxiv_id":"1808.07903","repositories_listed":4,"syntology":null},{"url":"/paper/bohb-robust-and-efficient-hyperparameter","slug":"bohb-robust-and-efficient-hyperparameter","title":"BOHB: Robust and Efficient Hyperparameter Optimization at Scale","date":"2018-07-04","arxiv_id":"1807.01774","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-playing-25d","slug":"deep-reinforcement-learning-for-playing-25d","title":"Deep Reinforcement Learning for Playing 2.5D Fighting Games","date":"2018-05-05","arxiv_id":"1805.02070","repositories_listed":4,"syntology":null},{"url":"/paper/learning-to-navigate-in-cities-without-a-map","slug":"learning-to-navigate-in-cities-without-a-map","title":"Learning to Navigate in Cities Without a Map","date":"2018-03-31","arxiv_id":"1804.00168","repositories_listed":4,"syntology":null},{"url":"/paper/learning-synergies-between-pushing-and","slug":"learning-synergies-between-pushing-and","title":"Learning Synergies between Pushing and Grasping with Self-supervised Deep Reinforcement Learning","date":"2018-03-27","arxiv_id":"1803.09956","repositories_listed":4,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-synergies-between-pushing-and#ran","syntology_url":"https://syntology.ai/paper/1803.09956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09956"}},"official":{"repos":["andyzeng/visual-pushing-grasping"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-bayesian-bandits-showdown-an-empirical","slug":"deep-bayesian-bandits-showdown-an-empirical","title":"Deep Bayesian Bandits Showdown: An Empirical Comparison of Bayesian Deep Networks for Thompson Sampling","date":"2018-02-26","arxiv_id":"1802.09127","repositories_listed":4,"syntology":null},{"url":"/paper/diversity-is-all-you-need-learning-skills","slug":"diversity-is-all-you-need-learning-skills","title":"Diversity is All You Need: Learning Skills without a Reward Function","date":"2018-02-16","arxiv_id":"1802.06070","repositories_listed":4,"syntology":{"n":11,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diversity-is-all-you-need-learning-skills#ran","syntology_url":"https://syntology.ai/paper/1802.06070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.06070"}},"official":null}},{"url":"/paper/reinforcement-learning-for-solving-the","slug":"reinforcement-learning-for-solving-the","title":"Reinforcement Learning for Solving the Vehicle Routing Problem","date":"2018-02-12","arxiv_id":"1802.04240","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-solving-the#ran","syntology_url":"https://syntology.ai/paper/1802.04240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.04240"}},"official":null}},{"url":"/paper/whatever-does-not-kill-deep-reinforcement","slug":"whatever-does-not-kill-deep-reinforcement","title":"Whatever Does Not Kill Deep Reinforcement Learning, Makes It Stronger","date":"2017-12-23","arxiv_id":"1712.09344","repositories_listed":4,"syntology":null},{"url":"/paper/ray-a-distributed-framework-for-emerging-ai","slug":"ray-a-distributed-framework-for-emerging-ai","title":"Ray: A Distributed Framework for Emerging AI Applications","date":"2017-12-16","arxiv_id":"1712.05889","repositories_listed":4,"syntology":null},{"url":"/paper/a-deeper-look-at-experience-replay","slug":"a-deeper-look-at-experience-replay","title":"A Deeper Look at Experience Replay","date":"2017-12-04","arxiv_id":"1712.01275","repositories_listed":4,"syntology":null},{"url":"/paper/learning-multi-level-hierarchies-with","slug":"learning-multi-level-hierarchies-with","title":"Learning Multi-Level Hierarchies with Hindsight","date":"2017-12-04","arxiv_id":"1712.00948","repositories_listed":4,"syntology":null},{"url":"/paper/embodied-question-answering","slug":"embodied-question-answering","title":"Embodied Question Answering","date":"2017-11-30","arxiv_id":"1711.11543","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-that-matters","slug":"deep-reinforcement-learning-that-matters","title":"Deep Reinforcement Learning that Matters","date":"2017-09-19","arxiv_id":"1709.06560","repositories_listed":4,"syntology":null},{"url":"/paper/leveraging-demonstrations-for-deep","slug":"leveraging-demonstrations-for-deep","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","date":"2017-07-27","arxiv_id":"1707.08817","repositories_listed":4,"syntology":null},{"url":"/paper/a-multi-agent-reinforcement-learning-model-of","slug":"a-multi-agent-reinforcement-learning-model-of","title":"A multi-agent reinforcement learning model of common-pool resource appropriation","date":"2017-07-20","arxiv_id":"1707.06600","repositories_listed":4,"syntology":null},{"url":"/paper/a-simple-neural-attentive-meta-learner","slug":"a-simple-neural-attentive-meta-learner","title":"A Simple Neural Attentive Meta-Learner","date":"2017-07-11","arxiv_id":"1707.03141","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/a-simple-neural-attentive-meta-learner#ran","syntology_url":"https://syntology.ai/paper/1707.03141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.03141"}},"official":null}},{"url":"/paper/thinking-fast-and-slow-with-deep-learning-and","slug":"thinking-fast-and-slow-with-deep-learning-and","title":"Thinking Fast and Slow with Deep Learning and Tree Search","date":"2017-05-23","arxiv_id":"1705.08439","repositories_listed":4,"syntology":null},{"url":"/paper/machine-comprehension-by-text-to-text-neural","slug":"machine-comprehension-by-text-to-text-neural","title":"Machine Comprehension by Text-to-Text Neural Question Generation","date":"2017-05-04","arxiv_id":"1705.02012","repositories_listed":4,"syntology":null},{"url":"/paper/neural-episodic-control","slug":"neural-episodic-control","title":"Neural Episodic Control","date":"2017-03-06","arxiv_id":"1703.01988","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-episodic-control#ran","syntology_url":"https://syntology.ai/paper/1703.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.01988"}},"official":null}},{"url":"/paper/reinforcement-learning-with-deep-energy-based","slug":"reinforcement-learning-with-deep-energy-based","title":"Reinforcement Learning with Deep Energy-Based Policies","date":"2017-02-27","arxiv_id":"1702.08165","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-with-deep-energy-based#ran","syntology_url":"https://syntology.ai/paper/1702.08165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1702.08165"}},"official":{"repos":["haarnoja/softqlearning"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/multi-agent-reinforcement-learning-in","slug":"multi-agent-reinforcement-learning-in","title":"Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2017-02-10","arxiv_id":"1702.03037","repositories_listed":4,"syntology":null},{"url":"/paper/dsac-differentiable-ransac-for-camera","slug":"dsac-differentiable-ransac-for-camera","title":"DSAC - Differentiable RANSAC for Camera Localization","date":"2016-11-17","arxiv_id":"1611.05705","repositories_listed":4,"syntology":null},{"url":"/paper/hierarchical-deep-reinforcement-learning","slug":"hierarchical-deep-reinforcement-learning","title":"Hierarchical Deep Reinforcement Learning: Integrating Temporal Abstraction and Intrinsic Motivation","date":"2016-04-20","arxiv_id":"1604.06057","repositories_listed":4,"syntology":null},{"url":"/paper/mastering-2048-with-delayed-temporal","slug":"mastering-2048-with-delayed-temporal","title":"Mastering 2048 with Delayed Temporal Coherence Learning, Multi-Stage Weight Promotion, Redundant Encoding and Carousel Shaping","date":"2016-04-18","arxiv_id":"1604.05085","repositories_listed":4,"syntology":null},{"url":"/paper/guided-cost-learning-deep-inverse-optimal","slug":"guided-cost-learning-deep-inverse-optimal","title":"Guided Cost Learning: Deep Inverse Optimal Control via Policy Optimization","date":"2016-03-01","arxiv_id":"1603.00448","repositories_listed":4,"syntology":null},{"url":"/paper/multiagent-cooperation-and-competition-with","slug":"multiagent-cooperation-and-competition-with","title":"Multiagent Cooperation and Competition with Deep Reinforcement Learning","date":"2015-11-27","arxiv_id":"1511.08779","repositories_listed":4,"syntology":null},{"url":"/paper/visionreasoner-unified-visual-perception-and","slug":"visionreasoner-unified-visual-perception-and","title":"VisionReasoner: Unified Visual Perception and Reasoning via Reinforcement Learning","date":"2025-05-17","arxiv_id":"2505.12081","repositories_listed":3,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visionreasoner-unified-visual-perception-and#ran","syntology_url":"https://syntology.ai/paper/2505.12081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12081"}},"official":{"repos":["dvlab-research/VisionReasoner","hiyouga/easyr1","dvlab-research/Seg-Zero"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ttrl-test-time-reinforcement-learning","slug":"ttrl-test-time-reinforcement-learning","title":"TTRL: Test-Time Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.16084","repositories_listed":3,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ttrl-test-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.16084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16084"}},"official":{"repos":["prime-rl/ttrl","tsinghuac3i/awesome-rl-reasoning-recipes"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","slug":"kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","arxiv_id":"2501.12599","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/kimi-k1-5-scaling-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2501.12599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.12599"}},"official":null}},{"url":"/paper/bricksrl-a-platform-for-democratizing","slug":"bricksrl-a-platform-for-democratizing","title":"BricksRL: A Platform for Democratizing Robotics and Reinforcement Learning Research and Education with LEGO","date":"2024-06-25","arxiv_id":"2406.17490","repositories_listed":3,"syntology":null},{"url":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","repositories_listed":3,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rebel-reinforcement-learning-via-regressing#ran","syntology_url":"https://syntology.ai/paper/2404.16767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16767"}},"official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/induced-model-matching-how-restricted-models","slug":"induced-model-matching-how-restricted-models","title":"Induced Model Matching: How Restricted Models Can Help Larger Ones","date":"2024-02-19","arxiv_id":"2402.12513","repositories_listed":3,"syntology":null},{"url":"/paper/evolving-reservoirs-for-meta-reinforcement","slug":"evolving-reservoirs-for-meta-reinforcement","title":"Evolving Reservoirs for Meta Reinforcement Learning","date":"2023-12-09","arxiv_id":"2312.06695","repositories_listed":3,"syntology":null},{"url":"/paper/rl4co-an-extensive-reinforcement-learning-for","slug":"rl4co-an-extensive-reinforcement-learning-for","title":"RL4CO: an Extensive Reinforcement Learning for Combinatorial Optimization Benchmark","date":"2023-06-29","arxiv_id":"2306.17100","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rl4co-an-extensive-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17100"}},"official":{"repos":["ai4co/rl4co","pytorch/rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","slug":"datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/datasets-and-benchmarks-for-offline-safe#ran","syntology_url":"https://syntology.ai/paper/2306.09303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09303"}},"official":{"repos":["liuzuxin/dsrl","liuzuxin/fsrl","liuzuxin/osrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/video-prediction-models-as-rewards-for","slug":"video-prediction-models-as-rewards-for","title":"Video Prediction Models as Rewards for Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14343","repositories_listed":3,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/video-prediction-models-as-rewards-for#ran","syntology_url":"https://syntology.ai/paper/2305.14343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14343"}},"official":null}},{"url":"/paper/training-diffusion-models-with-reinforcement","slug":"training-diffusion-models-with-reinforcement","title":"Training Diffusion Models with Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13301","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/training-diffusion-models-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2305.13301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13301"}},"official":{"repos":["kvablack/ddpo-pytorch"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/revisiting-the-minimalist-approach-to-offline","slug":"revisiting-the-minimalist-approach-to-offline","title":"Revisiting the Minimalist Approach to Offline Reinforcement Learning","date":"2023-05-16","arxiv_id":"2305.09836","repositories_listed":3,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-minimalist-approach-to-offline#ran","syntology_url":"https://syntology.ai/paper/2305.09836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09836"}},"official":{"repos":["dt6a/rebrac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/popgym-benchmarking-partially-observable","slug":"popgym-benchmarking-partially-observable","title":"POPGym: Benchmarking Partially Observable Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01859","repositories_listed":3,"syntology":null},{"url":"/paper/the-dormant-neuron-phenomenon-in-deep","slug":"the-dormant-neuron-phenomenon-in-deep","title":"The Dormant Neuron Phenomenon in Deep Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12902","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-dormant-neuron-phenomenon-in-deep#ran","syntology_url":"https://syntology.ai/paper/2302.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12902"}},"official":{"repos":["google/dopamine"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/optimizing-prompts-for-text-to-image-1","slug":"optimizing-prompts-for-text-to-image-1","title":"Optimizing Prompts for Text-to-Image Generation","date":"2022-12-19","arxiv_id":"2212.09611","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/optimizing-prompts-for-text-to-image-1#ran","syntology_url":"https://syntology.ai/paper/2212.09611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09611"}},"official":{"repos":["microsoft/lmops"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/in-context-reinforcement-learning-with","slug":"in-context-reinforcement-learning-with","title":"In-context Reinforcement Learning with Algorithm Distillation","date":"2022-10-25","arxiv_id":"2210.14215","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/in-context-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2210.14215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.14215"}},"official":null}},{"url":"/paper/model-based-offline-reinforcement-learning","slug":"model-based-offline-reinforcement-learning","title":"Model-Based Offline Reinforcement Learning with Pessimism-Modulated Dynamics Belief","date":"2022-10-13","arxiv_id":"2210.06692","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-agent-2","slug":"deep-reinforcement-learning-for-multi-agent-2","title":"Deep Reinforcement Learning for Multi-Agent Interaction","date":"2022-08-02","arxiv_id":"2208.01769","repositories_listed":3,"syntology":null}],"record_sha256":"27c7057c2279943d07bb9e6c0e75a20864bff14c9a2f8892489f8e622ef2bf65","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}