{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/openai-gym/papers/2","list_of":"/task/openai-gym","task":"OpenAI Gym","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":382,"counts":{"archive_papers_tagged":382,"with_a_code_link":179,"where_syntology_ran_a_sample":43,"not_listed_spam_title":0,"listed":382,"listed_where_code_ran":43,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":37,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":37,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/openai-gym","prev":"/task/openai-gym","next":"/task/openai-gym/papers/3","papers":[{"url":"/paper/catastrophic-interference-in-reinforcement","slug":"catastrophic-interference-in-reinforcement","title":"Catastrophic Interference in Reinforcement Learning: A Solution Based on Context Division and Knowledge Distillation","date":"2021-09-01","arxiv_id":"2109.00525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/catastrophic-interference-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.00525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00525"}},"official":{"repos":["sweety-dm/interference-aware-deep-q-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/constrained-policy-gradient-method-for-safe","slug":"constrained-policy-gradient-method-for-safe","title":"Constrained Policy Gradient Method for Safe and Fast Reinforcement Learning: a Neural Tangent Kernel Based Approach","date":"2021-07-19","arxiv_id":"2107.09139","repositories_listed":1,"syntology":null},{"url":"/paper/multi-goal-reinforcement-learning","slug":"multi-goal-reinforcement-learning","title":"Multi-Goal Reinforcement Learning environments for simulated Franka Emika Panda robot","date":"2021-06-25","arxiv_id":"2106.13687","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-goal-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2106.13687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13687"}},"official":{"repos":["qgallouedec/panda-gym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/brax-a-differentiable-physics-engine-for","slug":"brax-a-differentiable-physics-engine-for","title":"Brax -- A Differentiable Physics Engine for Large Scale Rigid Body Simulation","date":"2021-06-24","arxiv_id":"2106.13281","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/brax-a-differentiable-physics-engine-for#ran","syntology_url":"https://syntology.ai/paper/2106.13281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13281"}},"official":{"repos":["google/brax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-continuous-control-with-episodic","slug":"solving-continuous-control-with-episodic","title":"Solving Continuous Control with Episodic Memory","date":"2021-06-16","arxiv_id":"2106.08832","repositories_listed":1,"syntology":null},{"url":"/paper/rsoccer-a-framework-for-studying","slug":"rsoccer-a-framework-for-studying","title":"rSoccer: A Framework for Studying Reinforcement Learning in Small and Very Small Size Robot Soccer","date":"2021-06-15","arxiv_id":"2106.12895","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-sparse-training-for-deep","slug":"dynamic-sparse-training-for-deep","title":"Dynamic Sparse Training for Deep Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04217","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-sparse-training-for-deep#ran","syntology_url":"https://syntology.ai/paper/2106.04217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04217"}},"official":{"repos":["GhadaSokar/Dynamic-Sparse-Training-for-Deep-Reinforcement-Learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-environment-for-1","slug":"a-reinforcement-learning-environment-for-1","title":"A Reinforcement Learning Environment for Multi-Service UAV-enabled Wireless Systems","date":"2021-05-11","arxiv_id":"2105.05094","repositories_listed":1,"syntology":null},{"url":"/paper/lemgorl-an-open-source-benchmark-tool-to","slug":"lemgorl-an-open-source-benchmark-tool-to","title":"Towards Real-World Deployment of Reinforcement Learning for Traffic Signal Control","date":"2021-03-30","arxiv_id":"2103.16223","repositories_listed":1,"syntology":null},{"url":"/paper/policy-information-capacity-information","slug":"policy-information-capacity-information","title":"Policy Information Capacity: Information-Theoretic Measure for Task Complexity in Deep Reinforcement Learning","date":"2021-03-23","arxiv_id":"2103.12726","repositories_listed":1,"syntology":null},{"url":"/paper/the-ai-arena-a-framework-for-distributed","slug":"the-ai-arena-a-framework-for-distributed","title":"The AI Arena: A Framework for Distributed Multi-Agent Reinforcement Learning","date":"2021-03-09","arxiv_id":"2103.05737","repositories_listed":1,"syntology":null},{"url":"/paper/foresee-then-evaluate-decomposing-value","slug":"foresee-then-evaluate-decomposing-value","title":"Foresee then Evaluate: Decomposing Value Estimation with Latent Future Prediction","date":"2021-03-03","arxiv_id":"2103.02225","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/foresee-then-evaluate-decomposing-value#ran","syntology_url":"https://syntology.ai/paper/2103.02225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.02225"}},"official":{"repos":["bluecontra/AAAI2021-VDFP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/modular-deep-reinforcement-learning-for","slug":"modular-deep-reinforcement-learning-for","title":"Modular Deep Reinforcement Learning for Continuous Motion Planning with Temporal Logic","date":"2021-02-24","arxiv_id":"2102.12855","repositories_listed":1,"syntology":null},{"url":"/paper/optimism-is-all-you-need-model-based","slug":"optimism-is-all-you-need-model-based","title":"MobILE: Model-Based Imitation Learning From Observation Alone","date":"2021-02-22","arxiv_id":"2102.10769","repositories_listed":1,"syntology":null},{"url":"/paper/deluca-a-differentiable-control-library","slug":"deluca-a-differentiable-control-library","title":"Deluca -- A Differentiable Control Library: Environments, Methods, and Benchmarking","date":"2021-02-19","arxiv_id":"2102.09968","repositories_listed":1,"syntology":null},{"url":"/paper/sim-env-decoupling-openai-gym-environments","slug":"sim-env-decoupling-openai-gym-environments","title":"Sim-Env: Decoupling OpenAI Gym Environments from Simulation Models","date":"2021-02-19","arxiv_id":"2102.09824","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-for","slug":"explainable-reinforcement-learning-for","title":"Explainable Reinforcement Learning for Longitudinal Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/longicontrol-a-reinforcement-learning","slug":"longicontrol-a-reinforcement-learning","title":"LongiControl: A Reinforcement Learning Environment for Longitudinal Vehicle Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bf-a-language-for-general-purpose-neural","slug":"bf-a-language-for-general-purpose-neural","title":"BF++: a language for general-purpose program synthesis","date":"2021-01-23","arxiv_id":"2101.09571","repositories_listed":1,"syntology":null},{"url":"/paper/developing-an-openai-gym-compatible-framework","slug":"developing-an-openai-gym-compatible-framework","title":"Developing an OpenAI Gym-compatible framework and simulation environment for testing Deep Reinforcement Learning agents solving the Ambulance Location Problem","date":"2021-01-12","arxiv_id":"2101.04434","repositories_listed":1,"syntology":null},{"url":"/paper/faults-in-deep-reinforcement-learning","slug":"faults-in-deep-reinforcement-learning","title":"Faults in Deep Reinforcement Learning Programs: A Taxonomy and A Detection Approach","date":"2021-01-01","arxiv_id":"2101.00135","repositories_listed":1,"syntology":null},{"url":"/paper/citylearn-standardizing-research-in-multi","slug":"citylearn-standardizing-research-in-multi","title":"CityLearn: Standardizing Research in Multi-Agent Reinforcement Learning for Demand Response and Urban Energy Management","date":"2020-12-18","arxiv_id":"2012.10504","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-learning-of-interpretable","slug":"evolutionary-learning-of-interpretable","title":"Evolutionary learning of interpretable decision trees","date":"2020-12-14","arxiv_id":"2012.07723","repositories_listed":1,"syntology":null},{"url":"/paper/navrep-unsupervised-representations-for","slug":"navrep-unsupervised-representations-for","title":"NavRep: Unsupervised Representations for Reinforcement Learning of Robot Navigation in Dynamic Human Environments","date":"2020-12-08","arxiv_id":"2012.04406","repositories_listed":1,"syntology":null},{"url":"/paper/resolving-implicit-coordination-in-multi","slug":"resolving-implicit-coordination-in-multi","title":"Resolving Implicit Coordination in Multi-Agent Deep Reinforcement Learning with Deep Q-Networks & Game Theory","date":"2020-12-08","arxiv_id":"2012.09136","repositories_listed":1,"syntology":null},{"url":"/paper/acn-sim-an-open-source-simulator-for-data","slug":"acn-sim-an-open-source-simulator-for-data","title":"ACN-Sim: An Open-Source Simulator for Data-Driven Electric Vehicle Charging Research","date":"2020-12-04","arxiv_id":"2012.02809","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-maximum-entropy-inverse","slug":"revisiting-maximum-entropy-inverse","title":"Revisiting Maximum Entropy Inverse Reinforcement Learning: New Perspectives and Algorithms","date":"2020-12-01","arxiv_id":"2012.00889","repositories_listed":1,"syntology":null},{"url":"/paper/nlpgym-a-toolkit-for-evaluating-rl-agents-on","slug":"nlpgym-a-toolkit-for-evaluating-rl-agents-on","title":"NLPGym -- A toolkit for evaluating RL agents on Natural Language Processing Tasks","date":"2020-11-16","arxiv_id":"2011.08272","repositories_listed":1,"syntology":null},{"url":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tonic-a-deep-reinforcement-learning-library#ran","syntology_url":"https://syntology.ai/paper/2011.07537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07537"}},"official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/amortized-variational-deep-q-network","slug":"amortized-variational-deep-q-network","title":"Amortized Variational Deep Q Network","date":"2020-11-03","arxiv_id":"2011.01706","repositories_listed":1,"syntology":null},{"url":"/paper/control-with-adaptive-q-learning","slug":"control-with-adaptive-q-learning","title":"Control with adaptive Q-learning","date":"2020-11-03","arxiv_id":"2011.02141","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-population","slug":"deep-reinforcement-learning-with-population","title":"Deep Reinforcement Learning with Population-Coded Spiking Neural Network for Continuous Control","date":"2020-10-19","arxiv_id":"2010.09635","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-with-population#ran","syntology_url":"https://syntology.ai/paper/2010.09635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09635"}},"official":{"repos":["combra-lab/pop-spiking-deep-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/what-about-taking-policy-as-input-of-value-1","slug":"what-about-taking-policy-as-input-of-value-1","title":"What About Inputing Policy in Value Function: Policy Representation and Policy-extended Value Function Approximator","date":"2020-10-19","arxiv_id":"2010.09536","repositories_listed":1,"syntology":null},{"url":"/paper/grac-self-guided-and-self-regularized-actor","slug":"grac-self-guided-and-self-regularized-actor","title":"GRAC: Self-Guided and Self-Regularized Actor-Critic","date":"2020-09-18","arxiv_id":"2009.08973","repositories_listed":1,"syntology":null},{"url":"/paper/vacsim-learning-effective-strategies-for","slug":"vacsim-learning-effective-strategies-for","title":"VacSIM: Learning Effective Strategies for COVID-19 Vaccine Distribution using Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.06602","repositories_listed":1,"syntology":null},{"url":"/paper/optimality-based-analysis-of-xcsf-compaction","slug":"optimality-based-analysis-of-xcsf-compaction","title":"Optimality-based Analysis of XCSF Compaction in Discrete Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01476","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-model-based-stochastic-value-gradient","slug":"on-the-model-based-stochastic-value-gradient","title":"On the model-based stochastic value gradient for continuous reinforcement learning","date":"2020-08-28","arxiv_id":"2008.12775","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-deep-reinforcement-learning","slug":"integrating-deep-reinforcement-learning","title":"Integrating Deep Reinforcement Learning Networks with Health System Simulations","date":"2020-07-21","arxiv_id":"2008.07434","repositories_listed":1,"syntology":null},{"url":"/paper/otoworld-towards-learning-to-separate-by","slug":"otoworld-towards-learning-to-separate-by","title":"OtoWorld: Towards Learning to Separate by Learning to Move","date":"2020-07-12","arxiv_id":"2007.06123","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/otoworld-towards-learning-to-separate-by#ran","syntology_url":"https://syntology.ai/paper/2007.06123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06123"}},"official":{"repos":["pseeth/otoworld"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/experience-replay-with-likelihood-free","slug":"experience-replay-with-likelihood-free","title":"Experience Replay with Likelihood-free Importance Weights","date":"2020-06-23","arxiv_id":"2006.13169","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/experience-replay-with-likelihood-free#ran","syntology_url":"https://syntology.ai/paper/2006.13169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13169"}},"official":null}},{"url":"/paper/analyzing-reinforcement-learning-benchmarks","slug":"analyzing-reinforcement-learning-benchmarks","title":"Analyzing Reinforcement Learning Benchmarks with Random Weight Guessing","date":"2020-04-16","arxiv_id":"2004.07707","repositories_listed":1,"syntology":null},{"url":"/paper/neural-game-engine-accurate-learning","slug":"neural-game-engine-accurate-learning","title":"Neural Game Engine: Accurate learning of generalizable forward models from pixels","date":"2020-03-23","arxiv_id":"2003.10520","repositories_listed":1,"syntology":null},{"url":"/paper/state-only-imitation-with-transition-dynamics-1","slug":"state-only-imitation-with-transition-dynamics-1","title":"State-only Imitation with Transition Dynamics Mismatch","date":"2020-02-27","arxiv_id":"2002.11879","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/state-only-imitation-with-transition-dynamics-1#ran","syntology_url":"https://syntology.ai/paper/2002.11879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.11879"}},"official":{"repos":["tgangwani/RL-Indirect-imitation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discrete-action-on-policy-learning-with","slug":"discrete-action-on-policy-learning-with","title":"Discrete Action On-Policy Learning with Action-Value Critic","date":"2020-02-10","arxiv_id":"2002.03534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-action-on-policy-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.03534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03534"}},"official":{"repos":["yuguangyue/CARSM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/continuous-action-reinforcement-learning-for","slug":"continuous-action-reinforcement-learning-for","title":"Continuous-action Reinforcement Learning for Playing Racing Games: Comparing SPG to PPO","date":"2020-01-15","arxiv_id":"2001.05270","repositories_listed":1,"syntology":null},{"url":"/paper/blue-river-controls-a-toolkit-for","slug":"blue-river-controls-a-toolkit-for","title":"Blue River Controls: A toolkit for Reinforcement Learning Control Systems on Hardware","date":"2020-01-07","arxiv_id":"2001.02254","repositories_listed":1,"syntology":null},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-playstation-reinforcement-learning","slug":"the-playstation-reinforcement-learning","title":"The PlayStation Reinforcement Learning Environment (PSXLE)","date":"2019-12-12","arxiv_id":"1912.06101","repositories_listed":1,"syntology":null},{"url":"/paper/playing-games-in-the-dark-an-approach-for","slug":"playing-games-in-the-dark-an-approach-for","title":"Playing Games in the Dark: An approach for cross-modality transfer in reinforcement learning","date":"2019-11-28","arxiv_id":"1911.12851","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-games-in-the-dark-an-approach-for#ran","syntology_url":"https://syntology.ai/paper/1911.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12851"}},"official":null}},{"url":"/paper/gym-ignition-reproducible-robotic-simulations","slug":"gym-ignition-reproducible-robotic-simulations","title":"Gym-Ignition: Reproducible Robotic Simulations for Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.01715","repositories_listed":1,"syntology":null},{"url":"/paper/mvfst-rl-an-asynchronous-rl-framework-for","slug":"mvfst-rl-an-asynchronous-rl-framework-for","title":"MVFST-RL: An Asynchronous RL Framework for Congestion Control with Delayed Actions","date":"2019-10-09","arxiv_id":"1910.04054","repositories_listed":1,"syntology":null},{"url":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy","slug":"v-mpo-on-policy-maximum-a-posteriori-policy","title":"V-MPO: On-Policy Maximum a Posteriori Policy Optimization for Discrete and Continuous Control","date":"2019-09-26","arxiv_id":"1909.12238","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy#ran","syntology_url":"https://syntology.ai/paper/1909.12238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12238"}},"official":null}},{"url":"/paper/self-supervised-state-control-through","slug":"self-supervised-state-control-through","title":"Self-Supervised State-Control through Intrinsic Mutual Information Rewards","date":"2019-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-active-inference","slug":"demystifying-active-inference","title":"Active inference: demystified and compared","date":"2019-09-24","arxiv_id":"1909.10863","repositories_listed":1,"syntology":null},{"url":"/paper/invariant-transform-experience-replay","slug":"invariant-transform-experience-replay","title":"Invariant Transform Experience Replay: Data Augmentation for Deep Reinforcement Learning","date":"2019-09-24","arxiv_id":"1909.10707","repositories_listed":1,"syntology":null},{"url":"/paper/mdp-playground-meta-features-in-reinforcement","slug":"mdp-playground-meta-features-in-reinforcement","title":"MDP Playground: An Analysis and Debug Testbed for Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.07750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-playground-meta-features-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1909.07750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07750"}},"official":{"repos":["automl/mdp-playground"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proximal-distilled-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09807"}},"official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/provably-efficient-imitation-learning-from","slug":"provably-efficient-imitation-learning-from","title":"Provably Efficient Imitation Learning from Observation Alone","date":"2019-05-27","arxiv_id":"1905.10948","repositories_listed":1,"syntology":null},{"url":"/paper/deep-ordinal-reinforcement-learning","slug":"deep-ordinal-reinforcement-learning","title":"Deep Ordinal Reinforcement Learning","date":"2019-05-06","arxiv_id":"1905.02005","repositories_listed":1,"syntology":null},{"url":"/paper/gym-gazebo2-a-toolkit-for-reinforcement","slug":"gym-gazebo2-a-toolkit-for-reinforcement","title":"gym-gazebo2, a toolkit for reinforcement learning using ROS 2 and Gazebo","date":"2019-03-14","arxiv_id":"1903.06278","repositories_listed":1,"syntology":null},{"url":"/paper/deep-active-localization","slug":"deep-active-localization","title":"Deep Active Localization","date":"2019-03-05","arxiv_id":"1903.01669","repositories_listed":1,"syntology":null},{"url":"/paper/flappy-hummingbird-an-open-source-dynamic","slug":"flappy-hummingbird-an-open-source-dynamic","title":"Flappy Hummingbird: An Open Source Dynamic Simulation of Flapping Wing Robots and Animals","date":"2019-02-25","arxiv_id":"1902.09628","repositories_listed":1,"syntology":null},{"url":"/paper/prolonets-neural-encoding-human-experts","slug":"prolonets-neural-encoding-human-experts","title":"Neural-encoding Human Experts' Domain Knowledge to Warm Start Reinforcement Learning","date":"2019-02-15","arxiv_id":"1902.06007","repositories_listed":1,"syntology":null},{"url":"/paper/deconfounding-reinforcement-learning-in","slug":"deconfounding-reinforcement-learning-in","title":"Deconfounding Reinforcement Learning in Observational Settings","date":"2018-12-26","arxiv_id":"1812.10576","repositories_listed":1,"syntology":null},{"url":"/paper/iroko-a-framework-to-prototype-reinforcement","slug":"iroko-a-framework-to-prototype-reinforcement","title":"Iroko: A Framework to Prototype Reinforcement Learning for Data Center Traffic Control","date":"2018-12-24","arxiv_id":"1812.09975","repositories_listed":1,"syntology":null},{"url":"/paper/relative-entropy-regularized-policy-iteration","slug":"relative-entropy-regularized-policy-iteration","title":"Relative Entropy Regularized Policy Iteration","date":"2018-12-05","arxiv_id":"1812.02256","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-improving-agent","slug":"reinforcement-learning-for-improving-agent","title":"Reinforcement Learning for Improving Agent Design","date":"2018-10-09","arxiv_id":"1810.03779","repositories_listed":1,"syntology":null},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/bindsnet-a-machine-learning-oriented-spiking","slug":"bindsnet-a-machine-learning-oriented-spiking","title":"BindsNET: A machine learning-oriented spiking neural networks library in Python","date":"2018-06-04","arxiv_id":"1806.01423","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-trainer-for-model-based","slug":"intelligent-trainer-for-model-based","title":"Intelligent Trainer for Model-Based Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09496","repositories_listed":1,"syntology":null},{"url":"/paper/advances-in-experience-replay","slug":"advances-in-experience-replay","title":"Advances in Experience Replay","date":"2018-05-15","arxiv_id":"1805.05536","repositories_listed":1,"syntology":null},{"url":"/paper/gan-q-learning","slug":"gan-q-learning","title":"GAN Q-learning","date":"2018-05-13","arxiv_id":"1805.04874","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-ddpg-method-with-prioritized","slug":"a-novel-ddpg-method-with-prioritized","title":"A novel DDPG method with prioritized experience replay","date":"2017-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mdp-environments-for-the-openai-gym","slug":"mdp-environments-for-the-openai-gym","title":"MDP environments for the OpenAI Gym","date":"2017-09-26","arxiv_id":"1709.09069","repositories_listed":1,"syntology":null},{"url":"/paper/benchmark-environments-for-multitask-learning","slug":"benchmark-environments-for-multitask-learning","title":"Benchmark Environments for Multitask Learning in Continuous Domains","date":"2017-08-14","arxiv_id":"1708.04352","repositories_listed":1,"syntology":null},{"url":"/paper/aixijs-a-software-demo-for-general","slug":"aixijs-a-software-demo-for-general","title":"AIXIjs: A Software Demo for General Reinforcement Learning","date":"2017-05-22","arxiv_id":"1705.07615","repositories_listed":1,"syntology":null},{"url":"/paper/beating-atari-with-natural-language-guided","slug":"beating-atari-with-natural-language-guided","title":"Beating Atari with Natural Language Guided Reinforcement Learning","date":"2017-04-18","arxiv_id":"1704.05539","repositories_listed":1,"syntology":null},{"url":"/paper/towards-generalization-and-simplicity-in","slug":"towards-generalization-and-simplicity-in","title":"Towards Generalization and Simplicity in Continuous Control","date":"2017-03-08","arxiv_id":"1703.02660","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-deep-reinforcement-learning","slug":"collaborative-deep-reinforcement-learning","title":"Collaborative Deep Reinforcement Learning","date":"2017-02-19","arxiv_id":"1702.05796","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-plasticity-loss-in-continual","slug":"mitigating-plasticity-loss-in-continual","title":"Mitigating Plasticity Loss in Continual Reinforcement Learning by Reducing Churn","date":"2025-05-31","arxiv_id":"2506.00592","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/mitigating-plasticity-loss-in-continual#ran","syntology_url":"https://syntology.ai/paper/2506.00592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00592"}},"official":null}},{"url":null,"slug":"stitch-ope-trajectory-stitching-with-guided","title":"STITCH-OPE: Trajectory Stitching with Guided Diffusion for Off-Policy Evaluation","date":"2025-05-27","arxiv_id":"2505.20781","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10992","title":"ReaCritic: Large Reasoning Transformer-based DRL Critic-model Scaling For Heterogeneous Networks","date":"2025-05-16","arxiv_id":"2505.10992","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-ril-interleaved-reinforcement-and","title":"IN-RIL: Interleaved Reinforcement and Imitation Learning for Policy Fine-Tuning","date":"2025-05-15","arxiv_id":"2505.10442","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-2d-1-packing-in-constrained","title":"Optimizing 2D+1 Packing in Constrained Environments Using Deep Reinforcement Learning","date":"2025-03-21","arxiv_id":"2503.17573","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-cost-real-world-implementation-of-the","title":"Low-cost Real-world Implementation of the Swing-up Pendulum for Deep Reinforcement Learning Experiments","date":"2025-03-14","arxiv_id":"2503.11065","repositories_listed":0,"syntology":null},{"url":null,"slug":"illuminating-spaces-deep-reinforcement","title":"Illuminating Spaces: Deep Reinforcement Learning and Laser-Wall Partitioning for Architectural Layout Generation","date":"2025-02-06","arxiv_id":"2502.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-deep-rl-scales-predictably","title":"Value-Based Deep RL Scales Predictably","date":"2025-02-06","arxiv_id":"2502.04327","repositories_listed":0,"syntology":null},{"url":null,"slug":"session-level-dynamic-ad-load-optimization","title":"Session-Level Dynamic Ad Load Optimization using Offline Robust Reinforcement Learning","date":"2025-01-09","arxiv_id":"2501.05591","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-evaluation-of-offline","title":"Robustness Evaluation of Offline Reinforcement Learning for Robot Control Against Action Perturbations","date":"2024-12-25","arxiv_id":"2412.18781","repositories_listed":0,"syntology":null},{"url":"/paper/stealing-that-free-lunch-exposing-the-limits","slug":"stealing-that-free-lunch-exposing-the-limits","title":"Stealing That Free Lunch: Exposing the Limits of Dyna-Style Reinforcement Learning","date":"2024-12-18","arxiv_id":"2412.14312","repositories_listed":0,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stealing-that-free-lunch-exposing-the-limits#ran","syntology_url":"https://syntology.ai/paper/2412.14312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14312"}},"official":null}},{"url":null,"slug":"optimizing-sensor-redundancy-in-sequential","title":"Optimizing Sensor Redundancy in Sequential Decision-Making Problems","date":"2024-12-10","arxiv_id":"2412.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-reinforcement-learning-testbed","title":"A Multi-Agent Reinforcement Learning Testbed for Cognitive Radio Applications","date":"2024-10-28","arxiv_id":"2410.21521","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-statistical-inference-for-time-varying","title":"Asymptotic Analysis of Sample-averaged Q-learning","date":"2024-10-14","arxiv_id":"2410.10737","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-data-and-calibrated-simulation","title":"The Smart Buildings Control Suite: A Diverse Open Source Benchmark to Evaluate and Scale HVAC Control Policies for Sustainability","date":"2024-10-02","arxiv_id":"2410.03756","repositories_listed":0,"syntology":null},{"url":null,"slug":"magics-adversarial-rl-with-minimax-actors","title":"MAGICS: Adversarial RL with Minimax Actors Guided by Implicit Critic Stackelberg for Convergent Neural Synthesis of Robot Safety","date":"2024-09-20","arxiv_id":"2409.13867","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01510","title":"Adaptive Planning with Generative Models under Uncertainty","date":"2024-08-02","arxiv_id":"2408.01510","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-hardware-fault-tolerance-in","title":"Enhancing Hardware Fault Tolerance in Machines with Reinforcement Learning Policy Gradient Algorithms","date":"2024-07-21","arxiv_id":"2407.15283","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-guide-to-combining-r-and","title":"A Comprehensive Guide to Combining R and Python code for Data Science, Machine Learning and Reinforcement Learning","date":"2024-07-19","arxiv_id":"2407.14695","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-control-using-intelligent-timing-of","title":"Traffic control using intelligent timing of traffic lights with reinforcement learning technique and real-time processing of surveillance camera images","date":"2024-05-22","arxiv_id":"2405.13256","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-oab-off-policy-policy-gradient-method","title":"Off-OAB: Off-Policy Policy Gradient Method with Optimal Action-Dependent Baseline","date":"2024-05-04","arxiv_id":"2405.02572","repositories_listed":0,"syntology":null}],"record_sha256":"77a136d81eda6a3703a6657b22bb876a0a29f1bb1711b6fb703b750746903a42","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}