{"url":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","slug":"reinforcement-learning-1","description_markdown":"**Reinforcement Learning (RL)** involves training an agent to take actions in an environment to maximize a cumulative reward signal. The agent interacts with the environment and learns by receiving feedback in the form of rewards or punishments for its actions. The goal of reinforcement learning is to find the optimal policy or decision-making strategy that maximizes the long-term reward.","categories":[{"name":"Computer Code","url":"/area/computer-code"},{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Knowledge Base","url":"/area/knowledge-base"},{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":15113,"papers_with_code":4749,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":20,"subtasks":6,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/reinforcement-learning-on-procgen","slug":"reinforcement-learning-on-procgen","dataset":"ProcGen","dataset_url":"/dataset/procgen","rows_in_archive":2,"metrics":["Mean Normalized Performance"],"first_row_in_archive_order":{"model":"PPG","paper_title":"Phasic Policy Gradient","paper_url":"/paper/phasic-policy-gradient","paper_date":"2020-09-09","arxiv_id":"2009.04416","code_links":[{"title":"opendilab/DI-engine","url":"https://github.com/opendilab/DI-engine/blob/main/ding/policy/ppg.py"},{"title":"openai/phasic-policy-gradient","url":"https://github.com/openai/phasic-policy-gradient"},{"title":"jjccero/pbrl","url":"https://github.com/jjccero/pbrl/tree/master/pbrl/algorithms/ppg"}],"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":0}}},{"leaderboard":"/sota/reinforcement-learning-rl-on-1","slug":"reinforcement-learning-rl-on-1","dataset":".","dataset_url":"/dataset/gun-detection-dataset","rows_in_archive":1,"metrics":["0..5sec"],"first_row_in_archive_order":{"model":".","paper_title":"Multi-Goal Reinforcement Learning: Challenging Robotics Environments and Request for Research","paper_url":"/paper/multi-goal-reinforcement-learning-challenging","paper_date":"2018-02-26","arxiv_id":"1802.09464","code_links":[{"title":"DartEnv/dart-env","url":"https://github.com/DartEnv/dart-env"},{"title":"flowersteam/curious","url":"https://github.com/flowersteam/curious"},{"title":"vvanirudh/imitation-learning-gym","url":"https://github.com/vvanirudh/imitation-learning-gym"},{"title":"zubair-irshad/imitation_learning","url":"https://github.com/zubair-irshad/imitation_learning"},{"title":"LucasSilve/openAIgym","url":"https://github.com/LucasSilve/openAIgym"},{"title":"YanglanWang/classic_control","url":"https://github.com/YanglanWang/classic_control"},{"title":"YijiongLin/ITER_KER_GER","url":"https://github.com/YijiongLin/ITER_KER_GER"},{"title":"ozcell/gym_wmgds_ma","url":"https://github.com/ozcell/gym_wmgds_ma"},{"title":"mbrucker07/experiments","url":"https://github.com/mbrucker07/experiments"},{"title":"shriram2112/Reinforcement_learning_gym","url":"https://github.com/shriram2112/Reinforcement_learning_gym"},{"title":"tsinghua-rll/gym-gridworld","url":"https://github.com/tsinghua-rll/gym-gridworld"},{"title":"nohboogy/gym","url":"https://github.com/nohboogy/gym"},{"title":"Christopheraburns/openai-gym","url":"https://github.com/Christopheraburns/openai-gym"},{"title":"ligy2016/gym","url":"https://github.com/ligy2016/gym"},{"title":"1110589721/openia-gym","url":"https://github.com/1110589721/openia-gym"},{"title":"developer-smartwebzone/-reinforcement-learning-algorithms.","url":"https://github.com/developer-smartwebzone/-reinforcement-learning-algorithms."},{"title":"davidsonic/self_brewed_gym","url":"https://github.com/davidsonic/self_brewed_gym"},{"title":"gtrll/dartenv","url":"https://github.com/gtrll/dartenv"},{"title":"jturner65/Getup-DartEnv","url":"https://github.com/jturner65/Getup-DartEnv"},{"title":"09jvilla/CS234_gym","url":"https://github.com/09jvilla/CS234_gym"},{"title":"Steve--Hunter/gym","url":"https://github.com/Steve--Hunter/gym"},{"title":"abhiksingla/gym","url":"https://github.com/abhiksingla/gym"},{"title":"tjcdev/mlpgym","url":"https://github.com/tjcdev/mlpgym"},{"title":"nmsquared/CS7641-Assignment-4","url":"https://github.com/nmsquared/CS7641-Assignment-4"},{"title":"EndingCredits/gym","url":"https://github.com/EndingCredits/gym"},{"title":"shinian123/gym","url":"https://github.com/shinian123/gym"},{"title":"a-ozeki/openAI_gym","url":"https://github.com/a-ozeki/openAI_gym"},{"title":"varuncs2011/rl","url":"https://github.com/varuncs2011/rl"}],"syntology":null}}],"datasets":[{"url":"/dataset/procgen","name":"ProcGen","full_name":"","num_papers_in_archive":177},{"url":"/dataset/prm800k","name":"PRM800K","full_name":"","num_papers_in_archive":53},{"url":"/dataset/maniskill2","name":"ManiSkill2","full_name":"","num_papers_in_archive":40},{"url":"/dataset/smacv2","name":"SMACv2","full_name":"","num_papers_in_archive":30},{"url":"/dataset/v-d4rl","name":"V-D4RL","full_name":"","num_papers_in_archive":16},{"url":"/dataset/gun-detection-dataset","name":"Gun Detection Dataset","full_name":"","num_papers_in_archive":9},{"url":"/dataset/quality-diversity-benchmark-suite","name":"QDax","full_name":"","num_papers_in_archive":6},{"url":"/dataset/avalon","name":"Avalon","full_name":"","num_papers_in_archive":5},{"url":"/dataset/finrl-meta","name":"FinRL-Meta","full_name":"","num_papers_in_archive":5},{"url":"/dataset/civrealm","name":"CivRealm","full_name":"","num_papers_in_archive":4},{"url":"/dataset/popgym","name":"POPGym","full_name":"Partially Observable Process Gym","num_papers_in_archive":3},{"url":"/dataset/midgard","name":"MIDGARD","full_name":"MIDGARD","num_papers_in_archive":2},{"url":"/dataset/emmoe-100","name":"EMMOE-100","full_name":"","num_papers_in_archive":1},{"url":"/dataset/lilgym","name":"lilGym","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mikasa-robo-dataset","name":"MIKASA-Robo Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/netsecdata","name":"NetSecData","full_name":"","num_papers_in_archive":1},{"url":"/dataset/opin-pref","name":"opin-pref","full_name":"","num_papers_in_archive":1},{"url":"/dataset/pushworld","name":"PushWorld","full_name":"","num_papers_in_archive":1},{"url":"/dataset/roomenv-v1","name":"RoomEnv-v1","full_name":"The Room environment - v1","num_papers_in_archive":1},{"url":"/dataset/roomenv-v2","name":"RoomEnv-v2","full_name":"The Room environment - v2","num_papers_in_archive":1}],"subtasks":[{"url":"/task/3d-point-cloud-reinforcement-learning","name":"3D Point Cloud Reinforcement Learning"},{"url":"/task/multi-objective-reinforcement-learning","name":"Multi-Objective Reinforcement Learning"},{"url":"/task/off-policy-evaluation","name":"Off-policy evaluation"},{"url":"/task/roomenv-v0","name":"RoomEnv-v0"},{"url":"/task/roomenv-v1","name":"RoomEnv-v1"},{"url":"/task/roomenv-v2","name":"RoomEnv-v2"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":4749,"tagged_in_all":15113,"items":[{"url":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","repositories_listed":161,"syntology":{"n":306,"n_ran":158,"n_unverified":148,"n_pointer_only":163}},{"url":"/paper/playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","repositories_listed":112,"syntology":{"n":117,"n_ran":56,"n_unverified":61,"n_pointer_only":56}},{"url":"/paper/deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","repositories_listed":97,"syntology":{"n":106,"n_ran":55,"n_unverified":51,"n_pointer_only":57}},{"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","repositories_listed":86,"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}},{"url":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","repositories_listed":86,"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}},{"url":"/paper/model-agnostic-meta-learning-for-fast","title":"Model-Agnostic Meta-Learning for Fast Adaptation of Deep Networks","date":"2017-03-09","arxiv_id":"1703.03400","repositories_listed":85,"syntology":{"n":154,"n_ran":86,"n_unverified":68,"n_pointer_only":57}},{"url":"/paper/prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","arxiv_id":"1511.05952","repositories_listed":77,"syntology":{"n":111,"n_ran":78,"n_unverified":33,"n_pointer_only":43}},{"url":"/paper/dueling-network-architectures-for-deep","title":"Dueling Network Architectures for Deep Reinforcement Learning","date":"2015-11-20","arxiv_id":"1511.06581","repositories_listed":73,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/asynchronous-methods-for-deep-reinforcement","title":"Asynchronous Methods for Deep Reinforcement Learning","date":"2016-02-04","arxiv_id":"1602.01783","repositories_listed":70,"syntology":{"n":95,"n_ran":38,"n_unverified":57,"n_pointer_only":12}},{"url":"/paper/addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","repositories_listed":67,"syntology":{"n":36,"n_ran":9,"n_unverified":27,"n_pointer_only":20}},{"url":"/paper/mastering-chess-and-shogi-by-self-play-with-a","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","date":"2017-12-05","arxiv_id":"1712.01815","repositories_listed":62,"syntology":{"n":17,"n_ran":13,"n_unverified":4,"n_pointer_only":9}},{"url":"/paper/darts-differentiable-architecture-search","title":"DARTS: Differentiable Architecture Search","date":"2018-06-24","arxiv_id":"1806.09055","repositories_listed":59,"syntology":{"n":156,"n_ran":66,"n_unverified":90,"n_pointer_only":48}},{"url":"/paper/soft-actor-critic-algorithms-and-applications","title":"Soft Actor-Critic Algorithms and Applications","date":"2018-12-13","arxiv_id":"1812.05905","repositories_listed":52,"syntology":{"n":40,"n_ran":10,"n_unverified":30,"n_pointer_only":1}},{"url":"/paper/openai-gym","title":"OpenAI Gym","date":"2016-06-05","arxiv_id":"1606.01540","repositories_listed":45,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/weight-uncertainty-in-neural-networks","title":"Weight Uncertainty in Neural Networks","date":"2015-05-20","arxiv_id":"1505.05424","repositories_listed":38,"syntology":{"n":15,"n_ran":12,"n_unverified":3,"n_pointer_only":7}},{"url":"/paper/rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","repositories_listed":34,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/self-critical-sequence-training-for-image","title":"Self-critical Sequence Training for Image Captioning","date":"2016-12-02","arxiv_id":"1612.00563","repositories_listed":31,"syntology":{"n":13,"n_ran":8,"n_unverified":5,"n_pointer_only":3}},{"url":"/paper/a-deep-reinforcement-learning-framework-for","title":"A Deep Reinforcement Learning Framework for the Financial Portfolio Management Problem","date":"2017-06-30","arxiv_id":"1706.10059","repositories_listed":30,"syntology":null},{"url":"/paper/multi-goal-reinforcement-learning-challenging","title":"Multi-Goal Reinforcement Learning: Challenging Robotics Environments and Request for Research","date":"2018-02-26","arxiv_id":"1802.09464","repositories_listed":28,"syntology":null},{"url":"/paper/hindsight-experience-replay","title":"Hindsight Experience Replay","date":"2017-07-05","arxiv_id":"1707.01495","repositories_listed":28,"syntology":{"n":16,"n_ran":16,"n_unverified":0,"n_pointer_only":9}},{"url":"/paper/simple-random-search-provides-a-competitive","title":"Simple random search provides a competitive approach to reinforcement learning","date":"2018-03-19","arxiv_id":"1803.07055","repositories_listed":26,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":3}},{"url":"/paper/impala-scalable-distributed-deep-rl-with","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","date":"2018-02-05","arxiv_id":"1802.01561","repositories_listed":24,"syntology":{"n":34,"n_ran":16,"n_unverified":18,"n_pointer_only":3}},{"url":"/paper/the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/parlai-a-dialog-research-software-platform","title":"ParlAI: A Dialog Research Software Platform","date":"2017-05-18","arxiv_id":"1705.06476","repositories_listed":23,"syntology":null},{"url":"/paper/evolution-strategies-as-a-scalable","title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","date":"2017-03-10","arxiv_id":"1703.03864","repositories_listed":23,"syntology":{"n":29,"n_ran":7,"n_unverified":22,"n_pointer_only":1}},{"url":"/paper/seqgan-sequence-generative-adversarial-nets","title":"SeqGAN: Sequence Generative Adversarial Nets with Policy Gradient","date":"2016-09-18","arxiv_id":"1609.05473","repositories_listed":23,"syntology":{"n":21,"n_ran":10,"n_unverified":11,"n_pointer_only":13}},{"url":"/paper/exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","arxiv_id":"1810.12894","repositories_listed":22,"syntology":{"n":43,"n_ran":26,"n_unverified":17,"n_pointer_only":15}},{"url":"/paper/world-models","title":"World Models","date":"2018-03-27","arxiv_id":"1803.10122","repositories_listed":22,"syntology":{"n":38,"n_ran":6,"n_unverified":32,"n_pointer_only":3}},{"url":"/paper/a-distributional-perspective-on-reinforcement","title":"A Distributional Perspective on Reinforcement Learning","date":"2017-07-21","arxiv_id":"1707.06887","repositories_listed":22,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/dream-to-control-learning-behaviors-by-latent","title":"Dream to Control: Learning Behaviors by Latent Imagination","date":"2019-12-03","arxiv_id":"1912.01603","repositories_listed":21,"syntology":{"n":62,"n_ran":43,"n_unverified":19,"n_pointer_only":9}}],"syntology_records":27,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}