{"url":"/task/reinforcement-learning","name":"Reinforcement Learning","slug":"reinforcement-learning","description_markdown":null,"categories":[{"name":"Methodology","url":"/area/methodology"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":13178,"papers_with_code":4183,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/reinforcement-learning-on-iris","slug":"reinforcement-learning-on-iris","dataset":"iris","dataset_url":"/dataset/iris-1","rows_in_archive":1,"metrics":["10 Images, 4*4 Stitching, Exact Accuracy"],"first_row_in_archive_order":{"model":"。","paper_title":"Efficient training and design of photonic neural network through neuroevolution","paper_url":"/paper/efficient-training-and-design-of-photonic","paper_date":"2019-08-04","arxiv_id":"1908.08012","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/mujoco","name":"MuJoCo","full_name":"","num_papers_in_archive":1638},{"url":"/dataset/omniverse-isaac-gym","name":"Omniverse Isaac Gym","full_name":"","num_papers_in_archive":240},{"url":"/dataset/iris-1","name":"iris","full_name":"iris","num_papers_in_archive":20},{"url":"/dataset/sbdt-soccer","name":"Soccer","full_name":"ISSIA-CNR Soccer","num_papers_in_archive":17},{"url":"/dataset/tennis","name":"Tennis","full_name":"","num_papers_in_archive":14},{"url":"/dataset/basketball","name":"Basketball","full_name":"NUST-NBA181","num_papers_in_archive":12},{"url":"/dataset/atlantis","name":"ATLANTIS","full_name":"","num_papers_in_archive":2},{"url":"/dataset/hammer","name":"HAMMER","full_name":"","num_papers_in_archive":2},{"url":"/dataset/truthgen","name":"TruthGen","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":4183,"tagged_in_all":13178,"items":[{"url":"/paper/proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","arxiv_id":"1707.06347","repositories_listed":188,"syntology":{"n":176,"n_ran":99,"n_unverified":77,"n_pointer_only":94}},{"url":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","repositories_listed":161,"syntology":{"n":306,"n_ran":158,"n_unverified":148,"n_pointer_only":163}},{"url":"/paper/grad-cam-visual-explanations-from-deep","title":"Grad-CAM: Visual Explanations from Deep Networks via Gradient-based Localization","date":"2016-10-07","arxiv_id":"1610.02391","repositories_listed":126,"syntology":{"n":141,"n_ran":79,"n_unverified":62,"n_pointer_only":68}},{"url":"/paper/playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","repositories_listed":112,"syntology":{"n":117,"n_ran":56,"n_unverified":61,"n_pointer_only":56}},{"url":"/paper/deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","repositories_listed":97,"syntology":{"n":106,"n_ran":55,"n_unverified":51,"n_pointer_only":57}},{"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","repositories_listed":86,"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}},{"url":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","repositories_listed":86,"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}},{"url":"/paper/model-agnostic-meta-learning-for-fast","title":"Model-Agnostic Meta-Learning for Fast Adaptation of Deep Networks","date":"2017-03-09","arxiv_id":"1703.03400","repositories_listed":85,"syntology":{"n":154,"n_ran":86,"n_unverified":68,"n_pointer_only":57}},{"url":"/paper/prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","arxiv_id":"1511.05952","repositories_listed":77,"syntology":{"n":111,"n_ran":78,"n_unverified":33,"n_pointer_only":43}},{"url":"/paper/dueling-network-architectures-for-deep","title":"Dueling Network Architectures for Deep Reinforcement Learning","date":"2015-11-20","arxiv_id":"1511.06581","repositories_listed":73,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/asynchronous-methods-for-deep-reinforcement","title":"Asynchronous Methods for Deep Reinforcement Learning","date":"2016-02-04","arxiv_id":"1602.01783","repositories_listed":70,"syntology":{"n":95,"n_ran":38,"n_unverified":57,"n_pointer_only":12}},{"url":"/paper/addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","repositories_listed":67,"syntology":{"n":36,"n_ran":9,"n_unverified":27,"n_pointer_only":20}},{"url":"/paper/mastering-chess-and-shogi-by-self-play-with-a","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","date":"2017-12-05","arxiv_id":"1712.01815","repositories_listed":62,"syntology":{"n":17,"n_ran":13,"n_unverified":4,"n_pointer_only":9}},{"url":"/paper/darts-differentiable-architecture-search","title":"DARTS: Differentiable Architecture Search","date":"2018-06-24","arxiv_id":"1806.09055","repositories_listed":59,"syntology":{"n":156,"n_ran":66,"n_unverified":90,"n_pointer_only":48}},{"url":"/paper/unity-a-general-platform-for-intelligent","title":"Unity: A General Platform for Intelligent Agents","date":"2018-09-07","arxiv_id":"1809.02627","repositories_listed":55,"syntology":{"n":44,"n_ran":7,"n_unverified":37,"n_pointer_only":0}},{"url":"/paper/soft-actor-critic-algorithms-and-applications","title":"Soft Actor-Critic Algorithms and Applications","date":"2018-12-13","arxiv_id":"1812.05905","repositories_listed":52,"syntology":{"n":40,"n_ran":10,"n_unverified":30,"n_pointer_only":1}},{"url":"/paper/openai-gym","title":"OpenAI Gym","date":"2016-06-05","arxiv_id":"1606.01540","repositories_listed":45,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/weight-uncertainty-in-neural-networks","title":"Weight Uncertainty in Neural Networks","date":"2015-05-20","arxiv_id":"1505.05424","repositories_listed":38,"syntology":{"n":15,"n_ran":12,"n_unverified":3,"n_pointer_only":7}},{"url":"/paper/rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","repositories_listed":34,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/self-critical-sequence-training-for-image","title":"Self-critical Sequence Training for Image Captioning","date":"2016-12-02","arxiv_id":"1612.00563","repositories_listed":31,"syntology":{"n":13,"n_ran":8,"n_unverified":5,"n_pointer_only":3}},{"url":"/paper/a-deep-reinforcement-learning-framework-for","title":"A Deep Reinforcement Learning Framework for the Financial Portfolio Management Problem","date":"2017-06-30","arxiv_id":"1706.10059","repositories_listed":30,"syntology":null},{"url":"/paper/dropout-as-a-bayesian-approximation","title":"Dropout as a Bayesian Approximation: Representing Model Uncertainty in Deep Learning","date":"2015-06-06","arxiv_id":"1506.02142","repositories_listed":29,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/representation-learning-with-contrastive","title":"Representation Learning with Contrastive Predictive Coding","date":"2018-07-10","arxiv_id":"1807.03748","repositories_listed":28,"syntology":{"n":45,"n_ran":29,"n_unverified":16,"n_pointer_only":22}},{"url":"/paper/multi-goal-reinforcement-learning-challenging","title":"Multi-Goal Reinforcement Learning: Challenging Robotics Environments and Request for Research","date":"2018-02-26","arxiv_id":"1802.09464","repositories_listed":28,"syntology":null},{"url":"/paper/hindsight-experience-replay","title":"Hindsight Experience Replay","date":"2017-07-05","arxiv_id":"1707.01495","repositories_listed":28,"syntology":{"n":16,"n_ran":16,"n_unverified":0,"n_pointer_only":9}},{"url":"/paper/simple-random-search-provides-a-competitive","title":"Simple random search provides a competitive approach to reinforcement learning","date":"2018-03-19","arxiv_id":"1803.07055","repositories_listed":26,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":3}},{"url":"/paper/impala-scalable-distributed-deep-rl-with","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","date":"2018-02-05","arxiv_id":"1802.01561","repositories_listed":24,"syntology":{"n":34,"n_ran":16,"n_unverified":18,"n_pointer_only":3}},{"url":"/paper/the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/parlai-a-dialog-research-software-platform","title":"ParlAI: A Dialog Research Software Platform","date":"2017-05-18","arxiv_id":"1705.06476","repositories_listed":23,"syntology":null},{"url":"/paper/evolution-strategies-as-a-scalable","title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","date":"2017-03-10","arxiv_id":"1703.03864","repositories_listed":23,"syntology":{"n":29,"n_ran":7,"n_unverified":22,"n_pointer_only":1}}],"syntology_records":27,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}