{"url":"/task/q-learning","name":"Q-Learning","slug":"q-learning","description_markdown":"The goal of Q-learning is to learn a policy, which tells an agent what action to take under what circumstances.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Playing Atari with Deep Reinforcement Learning](https://arxiv.org/pdf/1312.5602v1.pdf) )</span>","categories":[{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1918,"papers_with_code":463,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/vizdoom","name":"VizDoom","full_name":"VizDoom","num_papers_in_archive":156},{"url":"/dataset/yeast","name":"Yeast","full_name":"","num_papers_in_archive":18}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":463,"tagged_in_all":1918,"items":[{"url":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","repositories_listed":161,"syntology":{"n":306,"n_ran":158,"n_unverified":148,"n_pointer_only":163}},{"url":"/paper/playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","repositories_listed":112,"syntology":{"n":117,"n_ran":56,"n_unverified":61,"n_pointer_only":56}},{"url":"/paper/deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","repositories_listed":97,"syntology":{"n":106,"n_ran":55,"n_unverified":51,"n_pointer_only":57}},{"url":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","repositories_listed":86,"syntology":{"n":148,"n_ran":91,"n_unverified":57,"n_pointer_only":66}},{"url":"/paper/multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","repositories_listed":86,"syntology":{"n":143,"n_ran":75,"n_unverified":68,"n_pointer_only":99}},{"url":"/paper/addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","repositories_listed":67,"syntology":{"n":36,"n_ran":9,"n_unverified":27,"n_pointer_only":20}},{"url":"/paper/evolution-strategies-as-a-scalable","title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","date":"2017-03-10","arxiv_id":"1703.03864","repositories_listed":23,"syntology":{"n":29,"n_ran":7,"n_unverified":22,"n_pointer_only":1}},{"url":"/paper/conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","arxiv_id":"2006.04779","repositories_listed":18,"syntology":{"n":34,"n_ran":24,"n_unverified":10,"n_pointer_only":5}},{"url":"/paper/offline-reinforcement-learning-with-implicit","title":"Offline Reinforcement Learning with Implicit Q-Learning","date":"2021-10-12","arxiv_id":"2110.06169","repositories_listed":17,"syntology":{"n":58,"n_ran":32,"n_unverified":26,"n_pointer_only":22}},{"url":"/paper/a-disembodied-developmental-robotic-agent","title":"A disembodied developmental robotic agent called Samu Bátfai","date":"2015-11-09","arxiv_id":"1511.02889","repositories_listed":15,"syntology":null},{"url":"/paper/deep-neuroevolution-genetic-algorithms-are-a","title":"Deep Neuroevolution: Genetic Algorithms Are a Competitive Alternative for Training Deep Neural Networks for Reinforcement Learning","date":"2017-12-18","arxiv_id":"1712.06567","repositories_listed":12,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/vizdoom-a-doom-based-ai-research-platform-for","title":"ViZDoom: A Doom-based AI Research Platform for Visual Reinforcement Learning","date":"2016-05-06","arxiv_id":"1605.02097","repositories_listed":10,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":5}},{"url":"/paper/rlpyt-a-research-code-base-for-deep","title":"rlpyt: A Research Code Base for Deep Reinforcement Learning in PyTorch","date":"2019-09-03","arxiv_id":"1909.01500","repositories_listed":9,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/continuous-deep-q-learning-with-model-based","title":"Continuous Deep Q-Learning with Model-based Acceleration","date":"2016-03-02","arxiv_id":"1603.00748","repositories_listed":8,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/optimization-of-molecules-via-deep","title":"Optimization of Molecules via Deep Reinforcement Learning","date":"2018-10-19","arxiv_id":"1810.08678","repositories_listed":7,"syntology":null},{"url":"/paper/playing-fps-games-with-deep-reinforcement","title":"Playing FPS Games with Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05521","repositories_listed":7,"syntology":null},{"url":"/paper/randomized-ensembled-double-q-learning-1","title":"Randomized Ensembled Double Q-Learning: Learning Fast Without a Model","date":"2021-01-15","arxiv_id":"2101.05982","repositories_listed":6,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/qplex-duplex-dueling-multi-agent-q-learning","title":"QPLEX: Duplex Dueling Multi-Agent Q-Learning","date":"2020-08-03","arxiv_id":"2008.01062","repositories_listed":6,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/deeptraffic-crowdsourced-hyperparameter","title":"DeepTraffic: Crowdsourced Hyperparameter Tuning of Deep Reinforcement Learning Systems for Multi-Agent Dense Traffic Navigation","date":"2018-01-09","arxiv_id":"1801.02805","repositories_listed":6,"syntology":null},{"url":"/paper/deep-q-learning-from-demonstrations","title":"Deep Q-learning from Demonstrations","date":"2017-04-12","arxiv_id":"1704.03732","repositories_listed":6,"syntology":null},{"url":"/paper/uncertainty-based-offline-reinforcement","title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","date":"2021-10-04","arxiv_id":"2110.01548","repositories_listed":5,"syntology":{"n":21,"n_ran":13,"n_unverified":8,"n_pointer_only":6}},{"url":"/paper/iq-learn-inverse-soft-q-learning-for","title":"IQ-Learn: Inverse soft-Q Learning for Imitation","date":"2021-06-23","arxiv_id":"2106.12142","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/benchmarking-batch-deep-reinforcement","title":"Benchmarking Batch Deep Reinforcement Learning Algorithms","date":"2019-10-03","arxiv_id":"1910.01708","repositories_listed":5,"syntology":null},{"url":"/paper/sqil-imitation-learning-via-regularized","title":"SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards","date":"2019-05-27","arxiv_id":"1905.11108","repositories_listed":5,"syntology":null},{"url":"/paper/stabilising-experience-replay-for-deep-multi","title":"Stabilising Experience Replay for Deep Multi-Agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08887","repositories_listed":5,"syntology":null},{"url":"/paper/designing-neural-network-architectures-using","title":"Designing Neural Network Architectures using Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.02167","repositories_listed":5,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/deep-recurrent-q-learning-for-partially","title":"Deep Recurrent Q-Learning for Partially Observable MDPs","date":"2015-07-23","arxiv_id":"1507.06527","repositories_listed":5,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/hierarchical-reinforcement-learning-with-the","title":"Hierarchical Reinforcement Learning with the MAXQ Value Function Decomposition","date":"1999-05-21","arxiv_id":"cs/9905014","repositories_listed":5,"syntology":null},{"url":"/paper/offline-rl-with-no-ood-actions-in-sample","title":"Offline RL with No OOD Actions: In-Sample Learning via Implicit Value Regularization","date":"2023-03-28","arxiv_id":"2303.15810","repositories_listed":4,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/extreme-q-learning-maxent-rl-without-entropy","title":"Extreme Q-Learning: MaxEnt RL without Entropy","date":"2023-01-05","arxiv_id":"2301.02328","repositories_listed":4,"syntology":{"n":13,"n_ran":8,"n_unverified":5,"n_pointer_only":7}}],"syntology_records":21,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}