{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/3","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":152,"rows_per_page":100,"rows":[201,300],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/2","next":"/task/reinforcement-learning-1/papers/4","papers":[{"url":"/paper/htmrl-biologically-plausible-reinforcement","slug":"htmrl-biologically-plausible-reinforcement","title":"HTMRL: Biologically Plausible Reinforcement Learning with Hierarchical Temporal Memory","date":"2020-09-18","arxiv_id":"2009.08880","repositories_listed":4,"syntology":null},{"url":"/paper/meta-learning-through-hebbian-plasticity-in","slug":"meta-learning-through-hebbian-plasticity-in","title":"Meta-Learning through Hebbian Plasticity in Random Networks","date":"2020-07-06","arxiv_id":"2007.02686","repositories_listed":4,"syntology":null},{"url":"/paper/sample-factory-egocentric-3d-control-from","slug":"sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-factory-egocentric-3d-control-from#ran","syntology_url":"https://syntology.ai/paper/2006.11751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11751"}},"official":{"repos":["alex-petrenko/sample-factory"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weighted-qmix-expanding-monotonic-value","slug":"weighted-qmix-expanding-monotonic-value","title":"Weighted QMIX: Expanding Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.10800","repositories_listed":4,"syntology":null},{"url":"/paper/deployment-efficient-reinforcement-learning","slug":"deployment-efficient-reinforcement-learning","title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","date":"2020-06-05","arxiv_id":"2006.03647","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deployment-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03647"}},"official":{"repos":["matsuolab/BREMEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-state-dependent-exploration-for#ran","syntology_url":"https://syntology.ai/paper/2005.05719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05719"}},"official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/planning-to-explore-via-self-supervised-world","slug":"planning-to-explore-via-self-supervised-world","title":"Planning to Explore via Self-Supervised World Models","date":"2020-05-12","arxiv_id":"2005.05960","repositories_listed":4,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/planning-to-explore-via-self-supervised-world#ran","syntology_url":"https://syntology.ai/paper/2005.05960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05960"}},"official":{"repos":["ramanans1/plan2explore"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/task-oriented-dialogue-system-for-automatic-1","slug":"task-oriented-dialogue-system-for-automatic-1","title":"Hierarchical Reinforcement Learning for Automatic Disease Diagnosis","date":"2020-04-29","arxiv_id":"2004.14254","repositories_listed":4,"syntology":null},{"url":"/paper/image-augmentation-is-all-you-need","slug":"image-augmentation-is-all-you-need","title":"Image Augmentation Is All You Need: Regularizing Deep Reinforcement Learning from Pixels","date":"2020-04-28","arxiv_id":"2004.13649","repositories_listed":4,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-augmentation-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2004.13649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.13649"}},"official":{"repos":["denisyarats/drq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","repositories_listed":4,"syntology":null},{"url":"/paper/discor-corrective-feedback-in-reinforcement","slug":"discor-corrective-feedback-in-reinforcement","title":"DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","date":"2020-03-16","arxiv_id":"2003.07305","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/discor-corrective-feedback-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.07305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07305"}},"official":null}},{"url":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/interpretable-end-to-end-urban-autonomous#ran","syntology_url":"https://syntology.ai/paper/2001.08726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08726"}},"official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/191202288","slug":"191202288","title":"Simplified Action Decoder for Deep Multi-Agent Reinforcement Learning","date":"2019-12-04","arxiv_id":"1912.02288","repositories_listed":4,"syntology":null},{"url":"/paper/comparing-observation-and-action","slug":"comparing-observation-and-action","title":"Comparing Observation and Action Representations for Deep Reinforcement Learning in $μ$RTS","date":"2019-10-26","arxiv_id":"1910.12134","repositories_listed":4,"syntology":null},{"url":"/paper/improving-sample-efficiency-in-model-free-1","slug":"improving-sample-efficiency-in-model-free-1","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","date":"2019-10-02","arxiv_id":"1910.01741","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-sample-efficiency-in-model-free-1#ran","syntology_url":"https://syntology.ai/paper/1910.01741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01741"}},"official":{"repos":["denisyarats/pytorch_sac_ae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/sme-net-sparse-motion-estimation-for","slug":"sme-net-sparse-motion-estimation-for","title":"SME-Net: Sparse Motion Estimation for Parametric Video Prediction Through Reinforcement Learning","date":"2019-10-01","arxiv_id":null,"repositories_listed":4,"syntology":null},{"url":"/paper/a-generalized-algorithm-for-multi-objective","slug":"a-generalized-algorithm-for-multi-objective","title":"A Generalized Algorithm for Multi-Objective Reinforcement Learning and Policy Adaptation","date":"2019-08-21","arxiv_id":"1908.08342","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-generalized-algorithm-for-multi-objective#ran","syntology_url":"https://syntology.ai/paper/1908.08342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.08342"}},"official":{"repos":["RunzheYang/MORL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/collaborative-multi-agent-dialogue-model","slug":"collaborative-multi-agent-dialogue-model","title":"Collaborative Multi-Agent Dialogue Model Training Via Reinforcement Learning","date":"2019-07-11","arxiv_id":"1907.05507","repositories_listed":4,"syntology":null},{"url":"/paper/qtran-learning-to-factorize-with","slug":"qtran-learning-to-factorize-with","title":"QTRAN: Learning to Factorize with Transformation for Cooperative Multi-Agent Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05408","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/qtran-learning-to-factorize-with#ran","syntology_url":"https://syntology.ai/paper/1905.05408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05408"}},"official":{"repos":["Sonkyunghwan/QTRAN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/multi-pass-q-networks-for-deep-reinforcement","slug":"multi-pass-q-networks-for-deep-reinforcement","title":"Multi-Pass Q-Networks for Deep Reinforcement Learning with Parameterised Action Spaces","date":"2019-05-10","arxiv_id":"1905.04388","repositories_listed":4,"syntology":null},{"url":"/paper/deep-neuroevolution-of-recurrent-and-discrete","slug":"deep-neuroevolution-of-recurrent-and-discrete","title":"Deep Neuroevolution of Recurrent and Discrete World Models","date":"2019-04-28","arxiv_id":"1906.08857","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-neuroevolution-of-recurrent-and-discrete#ran","syntology_url":"https://syntology.ai/paper/1906.08857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.08857"}},"official":{"repos":["sebastianrisi/ga-world-models"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/graph-convolutional-reinforcement-learning","slug":"graph-convolutional-reinforcement-learning","title":"Graph Convolutional Reinforcement Learning","date":"2018-10-22","arxiv_id":"1810.09202","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-convolutional-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.09202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09202"}},"official":{"repos":["PKU-AI-Edge/DGN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/lift-reinforcement-learning-in-computer","slug":"lift-reinforcement-learning-in-computer","title":"LIFT: Reinforcement Learning in Computer Systems by Learning From Demonstrations","date":"2018-08-23","arxiv_id":"1808.07903","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-playing-25d","slug":"deep-reinforcement-learning-for-playing-25d","title":"Deep Reinforcement Learning for Playing 2.5D Fighting Games","date":"2018-05-05","arxiv_id":"1805.02070","repositories_listed":4,"syntology":null},{"url":"/paper/learning-to-navigate-in-cities-without-a-map","slug":"learning-to-navigate-in-cities-without-a-map","title":"Learning to Navigate in Cities Without a Map","date":"2018-03-31","arxiv_id":"1804.00168","repositories_listed":4,"syntology":null},{"url":"/paper/learning-synergies-between-pushing-and","slug":"learning-synergies-between-pushing-and","title":"Learning Synergies between Pushing and Grasping with Self-supervised Deep Reinforcement Learning","date":"2018-03-27","arxiv_id":"1803.09956","repositories_listed":4,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-synergies-between-pushing-and#ran","syntology_url":"https://syntology.ai/paper/1803.09956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09956"}},"official":{"repos":["andyzeng/visual-pushing-grasping"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-bayesian-bandits-showdown-an-empirical","slug":"deep-bayesian-bandits-showdown-an-empirical","title":"Deep Bayesian Bandits Showdown: An Empirical Comparison of Bayesian Deep Networks for Thompson Sampling","date":"2018-02-26","arxiv_id":"1802.09127","repositories_listed":4,"syntology":null},{"url":"/paper/diversity-is-all-you-need-learning-skills","slug":"diversity-is-all-you-need-learning-skills","title":"Diversity is All You Need: Learning Skills without a Reward Function","date":"2018-02-16","arxiv_id":"1802.06070","repositories_listed":4,"syntology":{"n":11,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diversity-is-all-you-need-learning-skills#ran","syntology_url":"https://syntology.ai/paper/1802.06070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.06070"}},"official":null}},{"url":"/paper/reinforcement-learning-for-solving-the","slug":"reinforcement-learning-for-solving-the","title":"Reinforcement Learning for Solving the Vehicle Routing Problem","date":"2018-02-12","arxiv_id":"1802.04240","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-solving-the#ran","syntology_url":"https://syntology.ai/paper/1802.04240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.04240"}},"official":null}},{"url":"/paper/whatever-does-not-kill-deep-reinforcement","slug":"whatever-does-not-kill-deep-reinforcement","title":"Whatever Does Not Kill Deep Reinforcement Learning, Makes It Stronger","date":"2017-12-23","arxiv_id":"1712.09344","repositories_listed":4,"syntology":null},{"url":"/paper/ray-a-distributed-framework-for-emerging-ai","slug":"ray-a-distributed-framework-for-emerging-ai","title":"Ray: A Distributed Framework for Emerging AI Applications","date":"2017-12-16","arxiv_id":"1712.05889","repositories_listed":4,"syntology":null},{"url":"/paper/a-deeper-look-at-experience-replay","slug":"a-deeper-look-at-experience-replay","title":"A Deeper Look at Experience Replay","date":"2017-12-04","arxiv_id":"1712.01275","repositories_listed":4,"syntology":null},{"url":"/paper/embodied-question-answering","slug":"embodied-question-answering","title":"Embodied Question Answering","date":"2017-11-30","arxiv_id":"1711.11543","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-that-matters","slug":"deep-reinforcement-learning-that-matters","title":"Deep Reinforcement Learning that Matters","date":"2017-09-19","arxiv_id":"1709.06560","repositories_listed":4,"syntology":null},{"url":"/paper/leveraging-demonstrations-for-deep","slug":"leveraging-demonstrations-for-deep","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","date":"2017-07-27","arxiv_id":"1707.08817","repositories_listed":4,"syntology":null},{"url":"/paper/a-multi-agent-reinforcement-learning-model-of","slug":"a-multi-agent-reinforcement-learning-model-of","title":"A multi-agent reinforcement learning model of common-pool resource appropriation","date":"2017-07-20","arxiv_id":"1707.06600","repositories_listed":4,"syntology":null},{"url":"/paper/thinking-fast-and-slow-with-deep-learning-and","slug":"thinking-fast-and-slow-with-deep-learning-and","title":"Thinking Fast and Slow with Deep Learning and Tree Search","date":"2017-05-23","arxiv_id":"1705.08439","repositories_listed":4,"syntology":null},{"url":"/paper/machine-comprehension-by-text-to-text-neural","slug":"machine-comprehension-by-text-to-text-neural","title":"Machine Comprehension by Text-to-Text Neural Question Generation","date":"2017-05-04","arxiv_id":"1705.02012","repositories_listed":4,"syntology":null},{"url":"/paper/neural-episodic-control","slug":"neural-episodic-control","title":"Neural Episodic Control","date":"2017-03-06","arxiv_id":"1703.01988","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-episodic-control#ran","syntology_url":"https://syntology.ai/paper/1703.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.01988"}},"official":null}},{"url":"/paper/reinforcement-learning-with-deep-energy-based","slug":"reinforcement-learning-with-deep-energy-based","title":"Reinforcement Learning with Deep Energy-Based Policies","date":"2017-02-27","arxiv_id":"1702.08165","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reinforcement-learning-with-deep-energy-based#ran","syntology_url":"https://syntology.ai/paper/1702.08165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1702.08165"}},"official":{"repos":["haarnoja/softqlearning"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/multi-agent-reinforcement-learning-in","slug":"multi-agent-reinforcement-learning-in","title":"Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2017-02-10","arxiv_id":"1702.03037","repositories_listed":4,"syntology":null},{"url":"/paper/hierarchical-deep-reinforcement-learning","slug":"hierarchical-deep-reinforcement-learning","title":"Hierarchical Deep Reinforcement Learning: Integrating Temporal Abstraction and Intrinsic Motivation","date":"2016-04-20","arxiv_id":"1604.06057","repositories_listed":4,"syntology":null},{"url":"/paper/multiagent-cooperation-and-competition-with","slug":"multiagent-cooperation-and-competition-with","title":"Multiagent Cooperation and Competition with Deep Reinforcement Learning","date":"2015-11-27","arxiv_id":"1511.08779","repositories_listed":4,"syntology":null},{"url":"/paper/delving-into-rl-for-image-generation-with-cot","slug":"delving-into-rl-for-image-generation-with-cot","title":"Delving into RL for Image Generation with CoT: A Study on DPO vs. GRPO","date":"2025-05-22","arxiv_id":"2505.17017","repositories_listed":3,"syntology":null},{"url":"/paper/ttrl-test-time-reinforcement-learning","slug":"ttrl-test-time-reinforcement-learning","title":"TTRL: Test-Time Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.16084","repositories_listed":3,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ttrl-test-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.16084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16084"}},"official":{"repos":["prime-rl/ttrl","tsinghuac3i/awesome-rl-reasoning-recipes"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-swe-bench-a-multilingual-benchmark-for","slug":"multi-swe-bench-a-multilingual-benchmark-for","title":"Multi-SWE-bench: A Multilingual Benchmark for Issue Resolving","date":"2025-04-03","arxiv_id":"2504.02605","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-swe-bench-a-multilingual-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2504.02605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02605"}},"official":{"repos":["multi-swe-bench/experiments","multi-swe-bench/mopenhands","multi-swe-bench/multi-swe-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/search-r1-training-llms-to-reason-and","slug":"search-r1-training-llms-to-reason-and","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","date":"2025-03-12","arxiv_id":"2503.09516","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/search-r1-training-llms-to-reason-and#ran","syntology_url":"https://syntology.ai/paper/2503.09516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09516"}},"official":{"repos":["petergriffinjin/search-r1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","slug":"kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","arxiv_id":"2501.12599","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/kimi-k1-5-scaling-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2501.12599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.12599"}},"official":null}},{"url":"/paper/beyond-the-rainbow-high-performance-deep","slug":"beyond-the-rainbow-high-performance-deep","title":"Beyond The Rainbow: High Performance Deep Reinforcement Learning on a Desktop PC","date":"2024-11-06","arxiv_id":"2411.03820","repositories_listed":3,"syntology":{"n":25,"n_ran":16,"n_constructed":10,"n_ran_checked":13,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":21,"phrase":"16 ran (of which 10 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-the-rainbow-high-performance-deep#ran","syntology_url":"https://syntology.ai/paper/2411.03820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03820"}},"official":{"repos":["viptankz/btr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/balance-reward-and-safety-optimization-for","slug":"balance-reward-and-safety-optimization-for","title":"Balance Reward and Safety Optimization for Safe Reinforcement Learning: A Perspective of Gradient Manipulation","date":"2024-05-02","arxiv_id":"2405.01677","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/balance-reward-and-safety-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2405.01677","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01677"}},"official":{"repos":["saferl-lab/safety-mujoco","pku-alignment/omnisafe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rebel-reinforcement-learning-via-regressing","slug":"rebel-reinforcement-learning-via-regressing","title":"REBEL: Reinforcement Learning via Regressing Relative Rewards","date":"2024-04-25","arxiv_id":"2404.16767","repositories_listed":3,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rebel-reinforcement-learning-via-regressing#ran","syntology_url":"https://syntology.ai/paper/2404.16767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16767"}},"official":{"repos":["Owen-Oertell/rlcm","zhaolingao/rebel"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperagent-a-simple-scalable-efficient-and","slug":"hyperagent-a-simple-scalable-efficient-and","title":"Q-Star Meets Scalable Posterior Sampling: Bridging Theory and Practice via HyperAgent","date":"2024-02-05","arxiv_id":"2402.10228","repositories_listed":3,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hyperagent-a-simple-scalable-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2402.10228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10228"}},"official":{"repos":["szrlee/hyperagent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/evolving-reservoirs-for-meta-reinforcement","slug":"evolving-reservoirs-for-meta-reinforcement","title":"Evolving Reservoirs for Meta Reinforcement Learning","date":"2023-12-09","arxiv_id":"2312.06695","repositories_listed":3,"syntology":null},{"url":"/paper/jaxmarl-multi-agent-rl-environments-in-jax","slug":"jaxmarl-multi-agent-rl-environments-in-jax","title":"JaxMARL: Multi-Agent RL Environments and Algorithms in JAX","date":"2023-11-16","arxiv_id":"2311.10090","repositories_listed":3,"syntology":{"n":23,"n_ran":22,"n_constructed":0,"n_ran_checked":22,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":22,"n_pointer_only":0,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 0 honoured, 0 violated, 22 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jaxmarl-multi-agent-rl-environments-in-jax#ran","syntology_url":"https://syntology.ai/paper/2311.10090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10090"}},"official":{"repos":["flairox/jaxmarl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/td-mpc2-scalable-robust-world-models-for","slug":"td-mpc2-scalable-robust-world-models-for","title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","date":"2023-10-25","arxiv_id":"2310.16828","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":4,"n_ran_checked":6,"n_instrument":6,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"12 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/td-mpc2-scalable-robust-world-models-for#ran","syntology_url":"https://syntology.ai/paper/2310.16828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16828"}},"official":null}},{"url":"/paper/a-prefrontal-cortex-inspired-architecture-for","slug":"a-prefrontal-cortex-inspired-architecture-for","title":"Improving Planning with Large Language Models: A Modular Agentic Architecture","date":"2023-09-30","arxiv_id":"2310.00194","repositories_listed":3,"syntology":null},{"url":"/paper/rl4co-an-extensive-reinforcement-learning-for","slug":"rl4co-an-extensive-reinforcement-learning-for","title":"RL4CO: an Extensive Reinforcement Learning for Combinatorial Optimization Benchmark","date":"2023-06-29","arxiv_id":"2306.17100","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rl4co-an-extensive-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17100"}},"official":{"repos":["ai4co/rl4co","pytorch/rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","slug":"datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/datasets-and-benchmarks-for-offline-safe#ran","syntology_url":"https://syntology.ai/paper/2306.09303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09303"}},"official":{"repos":["liuzuxin/dsrl","liuzuxin/fsrl","liuzuxin/osrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/let-s-verify-step-by-step-1","slug":"let-s-verify-step-by-step-1","title":"Let's Verify Step by Step","date":"2023-05-31","arxiv_id":"2305.20050","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-s-verify-step-by-step-1#ran","syntology_url":"https://syntology.ai/paper/2305.20050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.20050"}},"official":{"repos":["openai/prm800k"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-minimalist-approach-to-offline","slug":"revisiting-the-minimalist-approach-to-offline","title":"Revisiting the Minimalist Approach to Offline Reinforcement Learning","date":"2023-05-16","arxiv_id":"2305.09836","repositories_listed":3,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-the-minimalist-approach-to-offline#ran","syntology_url":"https://syntology.ai/paper/2305.09836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.09836"}},"official":{"repos":["dt6a/rebrac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/contrastive-energy-prediction-for-exact","slug":"contrastive-energy-prediction-for-exact","title":"Contrastive Energy Prediction for Exact Energy-Guided Diffusion Sampling in Offline Reinforcement Learning","date":"2023-04-25","arxiv_id":"2304.12824","repositories_listed":3,"syntology":{"n":27,"n_ran":19,"n_constructed":1,"n_ran_checked":13,"n_instrument":6,"n_unverified":8,"n_honours":3,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"19 ran (of which 1 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/contrastive-energy-prediction-for-exact#ran","syntology_url":"https://syntology.ai/paper/2304.12824","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.12824"}},"official":{"repos":["thu-ml/cep-energy-guided-diffusion"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cal-ql-calibrated-offline-rl-pre-training-for-1","slug":"cal-ql-calibrated-offline-rl-pre-training-for-1","title":"Cal-QL: Calibrated Offline RL Pre-Training for Efficient Online Fine-Tuning","date":"2023-03-09","arxiv_id":"2303.05479","repositories_listed":3,"syntology":null},{"url":"/paper/popgym-benchmarking-partially-observable","slug":"popgym-benchmarking-partially-observable","title":"POPGym: Benchmarking Partially Observable Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01859","repositories_listed":3,"syntology":null},{"url":"/paper/the-dormant-neuron-phenomenon-in-deep","slug":"the-dormant-neuron-phenomenon-in-deep","title":"The Dormant Neuron Phenomenon in Deep Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12902","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-dormant-neuron-phenomenon-in-deep#ran","syntology_url":"https://syntology.ai/paper/2302.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12902"}},"official":{"repos":["google/dopamine"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/grounding-large-language-models-in","slug":"grounding-large-language-models-in","title":"Grounding Large Language Models in Interactive Environments with Online Reinforcement Learning","date":"2023-02-06","arxiv_id":"2302.02662","repositories_listed":3,"syntology":{"n":12,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/grounding-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2302.02662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02662"}},"official":{"repos":["clementromac/lamorel","flowersteam/grounding_llms_with_online_rl","flowersteam/lamorel"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-offline-reinforcement-learning","slug":"model-based-offline-reinforcement-learning","title":"Model-Based Offline Reinforcement Learning with Pessimism-Modulated Dynamics Belief","date":"2022-10-13","arxiv_id":"2210.06692","repositories_listed":3,"syntology":null},{"url":"/paper/is-reinforcement-learning-not-for-natural","slug":"is-reinforcement-learning-not-for-natural","title":"Is Reinforcement Learning (Not) for Natural Language Processing: Benchmarks, Baselines, and Building Blocks for Natural Language Policy Optimization","date":"2022-10-03","arxiv_id":"2210.01241","repositories_listed":3,"syntology":null},{"url":"/paper/learning-low-frequency-motion-control-for","slug":"learning-low-frequency-motion-control-for","title":"Learning Low-Frequency Motion Control for Robust and Dynamic Robot Locomotion","date":"2022-09-29","arxiv_id":"2209.14887","repositories_listed":3,"syntology":null},{"url":"/paper/constrained-update-projection-approach-to","slug":"constrained-update-projection-approach-to","title":"Constrained Update Projection Approach to Safe Policy Optimization","date":"2022-09-15","arxiv_id":"2209.07089","repositories_listed":3,"syntology":null},{"url":"/paper/diffusion-policies-as-an-expressive-policy","slug":"diffusion-policies-as-an-expressive-policy","title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","date":"2022-08-12","arxiv_id":"2208.06193","repositories_listed":3,"syntology":{"n":18,"n_ran":11,"n_constructed":5,"n_ran_checked":11,"n_instrument":0,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/diffusion-policies-as-an-expressive-policy#ran","syntology_url":"https://syntology.ai/paper/2208.06193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.06193"}},"official":{"repos":["zhendong-wang/diffusion-policies-for-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-for-multi-agent-2","slug":"deep-reinforcement-learning-for-multi-agent-2","title":"Deep Reinforcement Learning for Multi-Agent Interaction","date":"2022-08-02","arxiv_id":"2208.01769","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-turbulence","slug":"deep-reinforcement-learning-for-turbulence","title":"Deep Reinforcement Learning for Turbulence Modeling in Large Eddy Simulations","date":"2022-06-21","arxiv_id":"2206.11038","repositories_listed":3,"syntology":null},{"url":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envpool-a-highly-parallel-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2206.10558","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10558"}},"official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/a-unified-approach-to-reinforcement-learning","slug":"a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-approach-to-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05825"}},"official":{"repos":["deepmind/open_spiel"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/mildly-conservative-q-learning-for-offline","slug":"mildly-conservative-q-learning-for-offline","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04745","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":6,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":11,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mildly-conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.04745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04745"}},"official":{"repos":["dmksjfl/mcq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/evolving-curricula-with-regret-based","slug":"evolving-curricula-with-regret-based","title":"Evolving Curricula with Regret-Based Environment Design","date":"2022-03-02","arxiv_id":"2203.01302","repositories_listed":3,"syntology":null},{"url":"/paper/supported-policy-optimization-for-offline","slug":"supported-policy-optimization-for-offline","title":"Supported Policy Optimization for Offline Reinforcement Learning","date":"2022-02-13","arxiv_id":"2202.06239","repositories_listed":3,"syntology":null},{"url":"/paper/the-shapley-value-in-machine-learning","slug":"the-shapley-value-in-machine-learning","title":"The Shapley Value in Machine Learning","date":"2022-02-11","arxiv_id":"2202.05594","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-shapley-value-in-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2202.05594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.05594"}},"official":{"repos":["benedekrozemberczki/shapley"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarially-trained-actor-critic-for","slug":"adversarially-trained-actor-critic-for","title":"Adversarially Trained Actor Critic for Offline Reinforcement Learning","date":"2022-02-05","arxiv_id":"2202.02446","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarially-trained-actor-critic-for#ran","syntology_url":"https://syntology.ai/paper/2202.02446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.02446"}},"official":{"repos":["microsoft/atac"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/maximum-entropy-population-based-training-for-1","slug":"maximum-entropy-population-based-training-for-1","title":"Maximum Entropy Population-Based Training for Zero-Shot Human-AI Coordination","date":"2021-12-22","arxiv_id":"2112.11701","repositories_listed":3,"syntology":null},{"url":"/paper/a-closer-look-at-advantage-filtered","slug":"a-closer-look-at-advantage-filtered","title":"A Closer Look at Advantage-Filtered Behavioral Cloning in High-Noise Datasets","date":"2021-10-10","arxiv_id":"2110.04698","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-closer-look-at-advantage-filtered#ran","syntology_url":"https://syntology.ai/paper/2110.04698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04698"}},"official":{"repos":["jakegrigsby/cc-afbc","jakegrigsby/deep_control","jakegrigsby/super_sac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-sensory-neuron-as-a-transformer","slug":"the-sensory-neuron-as-a-transformer","title":"The Sensory Neuron as a Transformer: Permutation-Invariant Neural Networks for Reinforcement Learning","date":"2021-09-07","arxiv_id":"2109.02869","repositories_listed":3,"syntology":{"n":8,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-sensory-neuron-as-a-transformer#ran","syntology_url":"https://syntology.ai/paper/2109.02869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.02869"}},"official":null}},{"url":"/paper/warpdrive-extremely-fast-end-to-end-deep","slug":"warpdrive-extremely-fast-end-to-end-deep","title":"WarpDrive: Extremely Fast End-to-End Deep Multi-Agent Reinforcement Learning on a GPU","date":"2021-08-31","arxiv_id":"2108.13976","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/warpdrive-extremely-fast-end-to-end-deep#ran","syntology_url":"https://syntology.ai/paper/2108.13976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13976"}},"official":{"repos":["salesforce/warp-drive"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":4,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-at-the-edge-of#ran","syntology_url":"https://syntology.ai/paper/2108.13264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13264"}},"official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-urban-driving-by-imitating-a","slug":"end-to-end-urban-driving-by-imitating-a","title":"End-to-End Urban Driving by Imitating a Reinforcement Learning Coach","date":"2021-08-18","arxiv_id":"2108.08265","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-urban-driving-by-imitating-a#ran","syntology_url":"https://syntology.ai/paper/2108.08265","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.08265"}},"official":{"repos":["zhejz/carla-roach"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stabilizing-deep-q-learning-with-convnets-and","slug":"stabilizing-deep-q-learning-with-convnets-and","title":"Stabilizing Deep Q-Learning with ConvNets and Vision Transformers under Data Augmentation","date":"2021-07-01","arxiv_id":"2107.00644","repositories_listed":3,"syntology":{"n":20,"n_ran":17,"n_constructed":0,"n_ran_checked":15,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":14,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 1 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/stabilizing-deep-q-learning-with-convnets-and#ran","syntology_url":"https://syntology.ai/paper/2107.00644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00644"}},"official":{"repos":["nicklashansen/dmcontrol-generalization-benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/local-policy-search-with-bayesian","slug":"local-policy-search-with-bayesian","title":"Local policy search with Bayesian optimization","date":"2021-06-22","arxiv_id":"2106.11899","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/local-policy-search-with-bayesian#ran","syntology_url":"https://syntology.ai/paper/2106.11899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11899"}},"official":{"repos":["sarmueller/gibo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/androidenv-a-reinforcement-learning-platform","slug":"androidenv-a-reinforcement-learning-platform","title":"AndroidEnv: A Reinforcement Learning Platform for Android","date":"2021-05-27","arxiv_id":"2105.13231","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/androidenv-a-reinforcement-learning-platform#ran","syntology_url":"https://syntology.ai/paper/2105.13231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13231"}},"official":{"repos":["deepmind/android_env"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/feasible-actor-critic-constrained","slug":"feasible-actor-critic-constrained","title":"Feasible Actor-Critic: Constrained Reinforcement Learning for Ensuring Statewise Safety","date":"2021-05-22","arxiv_id":"2105.10682","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/feasible-actor-critic-constrained#ran","syntology_url":"https://syntology.ai/paper/2105.10682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10682"}},"official":{"repos":["mahaitongdae/Feasible-Actor-Critic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-multi-agent-reinforcement-learning-for-2","slug":"deep-multi-agent-reinforcement-learning-for-2","title":"Deep Multi-agent Reinforcement Learning for Highway On-Ramp Merging in Mixed Traffic","date":"2021-05-12","arxiv_id":"2105.05701","repositories_listed":3,"syntology":null},{"url":"/paper/constructions-in-combinatorics-via-neural","slug":"constructions-in-combinatorics-via-neural","title":"Constructions in combinatorics via neural networks","date":"2021-04-29","arxiv_id":"2104.14516","repositories_listed":3,"syntology":null},{"url":"/paper/podracer-architectures-for-scalable","slug":"podracer-architectures-for-scalable","title":"Podracer architectures for scalable Reinforcement Learning","date":"2021-04-13","arxiv_id":"2104.06272","repositories_listed":3,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/podracer-architectures-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2104.06272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06272"}},"official":null}},{"url":"/paper/integrated-decision-and-control-towards","slug":"integrated-decision-and-control-towards","title":"Integrated Decision and Control: Towards Interpretable and Computationally Efficient Driving Intelligence","date":"2021-03-18","arxiv_id":"2103.10290","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-fly-a-gym-environment-with","slug":"learning-to-fly-a-gym-environment-with","title":"Learning to Fly -- a Gym Environment with PyBullet Physics for Reinforcement Learning of Multi-agent Quadcopter Control","date":"2021-03-03","arxiv_id":"2103.02142","repositories_listed":3,"syntology":null},{"url":"/paper/near-real-world-benchmarks-for-offline","slug":"near-real-world-benchmarks-for-offline","title":"NeoRL: A Near Real-World Benchmark for Offline Reinforcement Learning","date":"2021-02-01","arxiv_id":"2102.00714","repositories_listed":3,"syntology":null},{"url":"/paper/meta-variationally-intrinsic-motivated","slug":"meta-variationally-intrinsic-motivated","title":"MetaVIM: Meta Variationally Intrinsic Motivated Reinforcement Learning for Decentralized Traffic Signal Control","date":"2021-01-04","arxiv_id":"2101.00746","repositories_listed":3,"syntology":null},{"url":"/paper/learning-fair-policies-in-decentralized","slug":"learning-fair-policies-in-decentralized","title":"Learning Fair Policies in Decentralized Cooperative Multi-Agent Reinforcement Learning","date":"2020-12-17","arxiv_id":"2012.09421","repositories_listed":3,"syntology":null},{"url":"/paper/generalization-in-reinforcement-learning-by","slug":"generalization-in-reinforcement-learning-by","title":"Generalization in Reinforcement Learning by Soft Data Augmentation","date":"2020-11-26","arxiv_id":"2011.13389","repositories_listed":3,"syntology":null},{"url":"/paper/pomo-policy-optimization-with-multiple-optima","slug":"pomo-policy-optimization-with-multiple-optima","title":"POMO: Policy Optimization with Multiple Optima for Reinforcement Learning","date":"2020-10-30","arxiv_id":"2010.16011","repositories_listed":3,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/pomo-policy-optimization-with-multiple-optima#ran","syntology_url":"https://syntology.ai/paper/2010.16011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.16011"}},"official":{"repos":["yd-kwon/POMO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-with-random-delays-1","slug":"reinforcement-learning-with-random-delays-1","title":"Reinforcement Learning with Random Delays","date":"2020-10-06","arxiv_id":"2010.02966","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-random-delays-1#ran","syntology_url":"https://syntology.ai/paper/2010.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.02966"}},"official":{"repos":["rmst/rlrd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"d354455532a479cf4ba2495eb2aa3367853833b3bff76092c6ab80e4d6c01da0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}