{"url":"/task/reinforcement-learning-2","name":"reinforcement-learning","slug":"reinforcement-learning-2","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":13427,"papers_with_code":4119,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/nasa-c-mapss","name":"NASA C-MAPSS","full_name":"Turbofan Engine Degradation Simulation Data Set","num_papers_in_archive":7}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":4119,"tagged_in_all":13427,"items":[{"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","repositories_listed":86,"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}},{"url":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","repositories_listed":86,"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}},{"url":"/paper/model-agnostic-meta-learning-for-fast","title":"Model-Agnostic Meta-Learning for Fast Adaptation of Deep Networks","date":"2017-03-09","arxiv_id":"1703.03400","repositories_listed":85,"syntology":{"n":154,"n_ran":86,"n_unverified":68,"n_pointer_only":57}},{"url":"/paper/prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","arxiv_id":"1511.05952","repositories_listed":77,"syntology":{"n":111,"n_ran":78,"n_unverified":33,"n_pointer_only":43}},{"url":"/paper/dueling-network-architectures-for-deep","title":"Dueling Network Architectures for Deep Reinforcement Learning","date":"2015-11-20","arxiv_id":"1511.06581","repositories_listed":73,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/asynchronous-methods-for-deep-reinforcement","title":"Asynchronous Methods for Deep Reinforcement Learning","date":"2016-02-04","arxiv_id":"1602.01783","repositories_listed":70,"syntology":{"n":95,"n_ran":38,"n_unverified":57,"n_pointer_only":12}},{"url":"/paper/addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","repositories_listed":67,"syntology":{"n":36,"n_ran":9,"n_unverified":27,"n_pointer_only":20}},{"url":"/paper/mastering-chess-and-shogi-by-self-play-with-a","title":"Mastering Chess and Shogi by Self-Play with a General Reinforcement Learning Algorithm","date":"2017-12-05","arxiv_id":"1712.01815","repositories_listed":62,"syntology":{"n":17,"n_ran":13,"n_unverified":4,"n_pointer_only":9}},{"url":"/paper/darts-differentiable-architecture-search","title":"DARTS: Differentiable Architecture Search","date":"2018-06-24","arxiv_id":"1806.09055","repositories_listed":59,"syntology":{"n":156,"n_ran":66,"n_unverified":90,"n_pointer_only":48}},{"url":"/paper/soft-actor-critic-algorithms-and-applications","title":"Soft Actor-Critic Algorithms and Applications","date":"2018-12-13","arxiv_id":"1812.05905","repositories_listed":52,"syntology":{"n":40,"n_ran":10,"n_unverified":30,"n_pointer_only":1}},{"url":"/paper/openai-gym","title":"OpenAI Gym","date":"2016-06-05","arxiv_id":"1606.01540","repositories_listed":45,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/weight-uncertainty-in-neural-networks","title":"Weight Uncertainty in Neural Networks","date":"2015-05-20","arxiv_id":"1505.05424","repositories_listed":38,"syntology":{"n":15,"n_ran":12,"n_unverified":3,"n_pointer_only":7}},{"url":"/paper/rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","repositories_listed":34,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/self-critical-sequence-training-for-image","title":"Self-critical Sequence Training for Image Captioning","date":"2016-12-02","arxiv_id":"1612.00563","repositories_listed":31,"syntology":{"n":13,"n_ran":8,"n_unverified":5,"n_pointer_only":3}},{"url":"/paper/a-deep-reinforcement-learning-framework-for","title":"A Deep Reinforcement Learning Framework for the Financial Portfolio Management Problem","date":"2017-06-30","arxiv_id":"1706.10059","repositories_listed":30,"syntology":null},{"url":"/paper/direct-preference-optimization-your-language","title":"Direct Preference Optimization: Your Language Model is Secretly a Reward Model","date":"2023-05-29","arxiv_id":"2305.18290","repositories_listed":29,"syntology":{"n":31,"n_ran":6,"n_unverified":25,"n_pointer_only":2}},{"url":"/paper/multi-goal-reinforcement-learning-challenging","title":"Multi-Goal Reinforcement Learning: Challenging Robotics Environments and Request for Research","date":"2018-02-26","arxiv_id":"1802.09464","repositories_listed":28,"syntology":null},{"url":"/paper/simple-random-search-provides-a-competitive","title":"Simple random search provides a competitive approach to reinforcement learning","date":"2018-03-19","arxiv_id":"1803.07055","repositories_listed":26,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":3}},{"url":"/paper/impala-scalable-distributed-deep-rl-with","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","date":"2018-02-05","arxiv_id":"1802.01561","repositories_listed":24,"syntology":{"n":34,"n_ran":16,"n_unverified":18,"n_pointer_only":3}},{"url":"/paper/parlai-a-dialog-research-software-platform","title":"ParlAI: A Dialog Research Software Platform","date":"2017-05-18","arxiv_id":"1705.06476","repositories_listed":23,"syntology":null},{"url":"/paper/evolution-strategies-as-a-scalable","title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","date":"2017-03-10","arxiv_id":"1703.03864","repositories_listed":23,"syntology":{"n":29,"n_ran":7,"n_unverified":22,"n_pointer_only":1}},{"url":"/paper/exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","arxiv_id":"1810.12894","repositories_listed":22,"syntology":{"n":43,"n_ran":26,"n_unverified":17,"n_pointer_only":15}},{"url":"/paper/world-models","title":"World Models","date":"2018-03-27","arxiv_id":"1803.10122","repositories_listed":22,"syntology":{"n":38,"n_ran":6,"n_unverified":32,"n_pointer_only":3}},{"url":"/paper/a-distributional-perspective-on-reinforcement","title":"A Distributional Perspective on Reinforcement Learning","date":"2017-07-21","arxiv_id":"1707.06887","repositories_listed":22,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/dream-to-control-learning-behaviors-by-latent","title":"Dream to Control: Learning Behaviors by Latent Imagination","date":"2019-12-03","arxiv_id":"1912.01603","repositories_listed":21,"syntology":{"n":62,"n_ran":43,"n_unverified":19,"n_pointer_only":9}},{"url":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","arxiv_id":"2106.01345","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_unverified":9,"n_pointer_only":6}},{"url":"/paper/the-surprising-effectiveness-of-mappo-in","title":"The Surprising Effectiveness of PPO in Cooperative, Multi-Agent Games","date":"2021-03-02","arxiv_id":"2103.01955","repositories_listed":19,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/implicit-quantile-networks-for-distributional","title":"Implicit Quantile Networks for Distributional Reinforcement Learning","date":"2018-06-14","arxiv_id":"1806.06923","repositories_listed":19,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/rl2-fast-reinforcement-learning-via-slow","title":"RL$^2$: Fast Reinforcement Learning via Slow Reinforcement Learning","date":"2016-11-09","arxiv_id":"1611.02779","repositories_listed":19,"syntology":{"n":21,"n_ran":11,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","arxiv_id":"2006.04779","repositories_listed":18,"syntology":{"n":34,"n_ran":24,"n_unverified":10,"n_pointer_only":5}}],"syntology_records":27,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}