{"url":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning","slug":"deep-reinforcement-learning","description_markdown":null,"categories":[{"name":"Methodology","url":"/area/methodology"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":5822,"papers_with_code":1739,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[{"url":"/task/reinforcement-learning","name":"Reinforcement Learning"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":1739,"tagged_in_all":5822,"items":[{"url":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","repositories_listed":161,"syntology":{"n":306,"n_ran":159,"n_unverified":147,"n_pointer_only":163}},{"url":"/paper/playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","repositories_listed":112,"syntology":{"n":117,"n_ran":56,"n_unverified":61,"n_pointer_only":56}},{"url":"/paper/deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","repositories_listed":97,"syntology":{"n":106,"n_ran":56,"n_unverified":50,"n_pointer_only":57}},{"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","repositories_listed":86,"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}},{"url":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","repositories_listed":86,"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}},{"url":"/paper/dueling-network-architectures-for-deep","title":"Dueling Network Architectures for Deep Reinforcement Learning","date":"2015-11-20","arxiv_id":"1511.06581","repositories_listed":73,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/asynchronous-methods-for-deep-reinforcement","title":"Asynchronous Methods for Deep Reinforcement Learning","date":"2016-02-04","arxiv_id":"1602.01783","repositories_listed":70,"syntology":{"n":95,"n_ran":39,"n_unverified":56,"n_pointer_only":12}},{"url":"/paper/rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","repositories_listed":34,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/a-deep-reinforcement-learning-framework-for","title":"A Deep Reinforcement Learning Framework for the Financial Portfolio Management Problem","date":"2017-06-30","arxiv_id":"1706.10059","repositories_listed":30,"syntology":null},{"url":"/paper/dropout-as-a-bayesian-approximation","title":"Dropout as a Bayesian Approximation: Representing Model Uncertainty in Deep Learning","date":"2015-06-06","arxiv_id":"1506.02142","repositories_listed":29,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","arxiv_id":"1810.12894","repositories_listed":22,"syntology":{"n":43,"n_ran":26,"n_unverified":17,"n_pointer_only":15}},{"url":"/paper/rl2-fast-reinforcement-learning-via-slow","title":"RL$^2$: Fast Reinforcement Learning via Slow Reinforcement Learning","date":"2016-11-09","arxiv_id":"1611.02779","repositories_listed":19,"syntology":{"n":21,"n_ran":11,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/gradient-surgery-for-multi-task-learning-1","title":"Gradient Surgery for Multi-Task Learning","date":"2020-01-19","arxiv_id":"2001.06782","repositories_listed":18,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/flow-architecture-and-benchmarking-for","title":"Flow: A Modular Learning Framework for Mixed Autonomy Traffic","date":"2017-10-16","arxiv_id":"1710.05465","repositories_listed":16,"syntology":{"n":17,"n_ran":1,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/distributed-prioritized-experience-replay","title":"Distributed Prioritized Experience Replay","date":"2018-03-02","arxiv_id":"1803.00933","repositories_listed":15,"syntology":{"n":15,"n_ran":0,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/noisy-networks-for-exploration","title":"Noisy Networks for Exploration","date":"2017-06-30","arxiv_id":"1706.10295","repositories_listed":15,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/benchmarking-deep-reinforcement-learning-for","title":"Benchmarking Deep Reinforcement Learning for Continuous Control","date":"2016-04-22","arxiv_id":"1604.06778","repositories_listed":15,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/dopamine-a-research-framework-for-deep","title":"Dopamine: A Research Framework for Deep Reinforcement Learning","date":"2018-12-14","arxiv_id":"1812.06110","repositories_listed":13,"syntology":{"n":16,"n_ran":0,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/deep-neuroevolution-genetic-algorithms-are-a","title":"Deep Neuroevolution: Genetic Algorithms Are a Competitive Alternative for Training Deep Neural Networks for Reinforcement Learning","date":"2017-12-18","arxiv_id":"1712.06567","repositories_listed":12,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/off-policy-deep-reinforcement-learning","title":"Off-Policy Deep Reinforcement Learning without Exploration","date":"2018-12-07","arxiv_id":"1812.02900","repositories_listed":10,"syntology":{"n":14,"n_ran":14,"n_unverified":0,"n_pointer_only":9}},{"url":"/paper/starcraft-ii-a-new-challenge-for","title":"StarCraft II: A New Challenge for Reinforcement Learning","date":"2017-08-16","arxiv_id":"1708.04782","repositories_listed":10,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/parameter-space-noise-for-exploration","title":"Parameter Space Noise for Exploration","date":"2017-06-06","arxiv_id":"1706.01905","repositories_listed":10,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/comparative-evaluation-of-multi-agent-deep","title":"Benchmarking Multi-Agent Deep Reinforcement Learning Algorithms in Cooperative Tasks","date":"2020-06-14","arxiv_id":"2006.07869","repositories_listed":9,"syntology":{"n":12,"n_ran":11,"n_unverified":1,"n_pointer_only":7}},{"url":"/paper/rlpyt-a-research-code-base-for-deep","title":"rlpyt: A Research Code Base for Deep Reinforcement Learning in PyTorch","date":"2019-09-03","arxiv_id":"1909.01500","repositories_listed":9,"syntology":{"n":10,"n_ran":5,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/stochastic-latent-actor-critic-deep","title":"Stochastic Latent Actor-Critic: Deep Reinforcement Learning with a Latent Variable Model","date":"2019-07-01","arxiv_id":"1907.00953","repositories_listed":9,"syntology":{"n":11,"n_ran":10,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/practical-deep-reinforcement-learning","title":"Practical Deep Reinforcement Learning Approach for Stock Trading","date":"2018-11-19","arxiv_id":"1811.07522","repositories_listed":9,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-a-handful-of","title":"Deep Reinforcement Learning in a Handful of Trials using Probabilistic Dynamics Models","date":"2018-05-30","arxiv_id":"1805.12114","repositories_listed":9,"syntology":{"n":19,"n_ran":12,"n_unverified":7,"n_pointer_only":17}},{"url":"/paper/solving-the-rubiks-cube-without-human","title":"Solving the Rubik's Cube Without Human Knowledge","date":"2018-05-18","arxiv_id":"1805.07470","repositories_listed":9,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/population-based-training-of-neural-networks","title":"Population Based Training of Neural Networks","date":"2017-11-27","arxiv_id":"1711.09846","repositories_listed":9,"syntology":null},{"url":"/paper/neural-network-dynamics-for-model-based-deep","title":"Neural Network Dynamics for Model-Based Deep Reinforcement Learning with Model-Free Fine-Tuning","date":"2017-08-08","arxiv_id":"1708.02596","repositories_listed":9,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}],"syntology_records":27,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}