{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/q-learning/papers/18","list_of":"/method/q-learning","method":"Q-Learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":18,"pages_in_order":18,"rows_per_page":100,"rows":[1701,1734],"of":1734,"counts":{"archive_papers_tagged":1734,"with_a_code_link":464,"where_syntology_ran_a_sample":126,"not_listed_spam_title":0,"listed":1734,"listed_where_code_ran":126,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":105,"every_run_a_failure_of_syntologys_instrument":21,"listed_with_a_run_with_no_instrument_failure":105,"listed_every_run_a_failure_of_syntologys_instrument":21,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/q-learning","prev":"/method/q-learning/papers/17","next":null,"papers":[{"paper":"/paper/deep-exploration-via-bootstrapped-dqn","slug":"deep-exploration-via-bootstrapped-dqn","title":"Deep Exploration via Bootstrapped DQN","date":"2016-02-15","arxiv_id":"1602.04621","n_code_links":6,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"using-deep-q-learning-to-control-optimization","title":"Using Deep Q-Learning to Control Optimization Hyperparameters","date":"2016-02-12","arxiv_id":"1602.04062","n_code_links":0,"syntology":null},{"paper":"/paper/angrier-birds-bayesian-reinforcement-learning","slug":"angrier-birds-bayesian-reinforcement-learning","title":"Angrier Birds: Bayesian reinforcement learning","date":"2016-01-06","arxiv_id":"1601.01297","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-to-discount-deep-reinforcement-learning","title":"How to Discount Deep Reinforcement Learning: Towards New Dynamic Strategies","date":"2015-12-07","arxiv_id":"1512.02011","n_code_links":0,"syntology":null},{"paper":"/paper/deep-attention-recurrent-q-network","slug":"deep-attention-recurrent-q-network","title":"Deep Attention Recurrent Q-Network","date":"2015-12-05","arxiv_id":"1512.01693","n_code_links":3,"syntology":null},{"paper":"/paper/state-of-the-art-control-of-atari-games-using","slug":"state-of-the-art-control-of-atari-games-using","title":"State of the Art Control of Atari Games Using Shallow Reinforcement Learning","date":"2015-12-04","arxiv_id":"1512.01563","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-attention","title":"Deep Reinforcement Learning with Attention for Slate Markov Decision Processes with High-Dimensional States and Actions","date":"2015-12-03","arxiv_id":"1512.01124","n_code_links":0,"syntology":null},{"paper":"/paper/multiagent-cooperation-and-competition-with","slug":"multiagent-cooperation-and-competition-with","title":"Multiagent Cooperation and Competition with Deep Reinforcement Learning","date":"2015-11-27","arxiv_id":"1511.08779","n_code_links":4,"syntology":null},{"paper":"/paper/policy-distillation","slug":"policy-distillation","title":"Policy Distillation","date":"2015-11-19","arxiv_id":"1511.06295","n_code_links":1,"syntology":null},{"paper":"/paper/prioritized-experience-replay","slug":"prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","arxiv_id":"1511.05952","n_code_links":77,"syntology":{"ran":80,"of":111,"n_ran_checked":72,"n_instrument":8,"unverified":31,"pointer_only":43,"phrase":"80 ran (of which 62 constructed an object rather than computing a result; 72 with no instrument failure: 4 honoured, 0 violated, 68 with no contract checked; 8 where Syntology's instrument failed) · 31 unverified","official":null}},{"paper":"/paper/deep-reinforcement-learning-with-a-natural","slug":"deep-reinforcement-learning-with-a-natural","title":"Deep Reinforcement Learning with a Natural Language Action Space","date":"2015-11-14","arxiv_id":"1511.04636","n_code_links":3,"syntology":null},{"paper":"/paper/a-disembodied-developmental-robotic-agent","slug":"a-disembodied-developmental-robotic-agent","title":"A disembodied developmental robotic agent called Samu Bátfai","date":"2015-11-09","arxiv_id":"1511.02889","n_code_links":15,"syntology":null},{"paper":null,"slug":"generating-text-with-deep-reinforcement","title":"Generating Text with Deep Reinforcement Learning","date":"2015-10-30","arxiv_id":"1510.09202","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-double-q","slug":"deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","n_code_links":97,"syntology":{"ran":56,"of":106,"n_ran_checked":55,"n_instrument":1,"unverified":50,"pointer_only":57,"phrase":"56 ran (of which 38 constructed an object rather than computing a result; 55 with no instrument failure: 0 honoured, 0 violated, 55 with no contract checked; 1 where Syntology's instrument failed) · 50 unverified","official":null}},{"paper":null,"slug":"optimization-of-anemia-treatment-in","title":"Optimization of anemia treatment in hemodialysis patients via reinforcement learning","date":"2015-09-14","arxiv_id":"1509.03977","n_code_links":0,"syntology":null},{"paper":"/paper/continuous-control-with-deep-reinforcement","slug":"continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","n_code_links":161,"syntology":{"ran":163,"of":306,"n_ran_checked":152,"n_instrument":11,"unverified":143,"pointer_only":163,"phrase":"163 ran (of which 126 constructed an object rather than computing a result; 152 with no instrument failure: 3 honoured, 0 violated, 149 with no contract checked; 11 where Syntology's instrument failed) · 143 unverified","official":null}},{"paper":null,"slug":"artificial-prediction-markets-for-online","title":"Artificial Prediction Markets for Online Prediction of Continuous Variables-A Preliminary Report","date":"2015-08-11","arxiv_id":"1508.02681","n_code_links":0,"syntology":null},{"paper":"/paper/massively-parallel-methods-for-deep","slug":"massively-parallel-methods-for-deep","title":"Massively Parallel Methods for Deep Reinforcement Learning","date":"2015-07-15","arxiv_id":"1507.04296","n_code_links":3,"syntology":null},{"paper":null,"slug":"online-transfer-learning-in-reinforcement","title":"Online Transfer Learning in Reinforcement Learning Domains","date":"2015-07-02","arxiv_id":"1507.00436","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomous-crm-control-via-clv-approximation","title":"Autonomous CRM Control via CLV Approximation with Deep Reinforcement Learning in Discrete and Continuous Action Space","date":"2015-04-08","arxiv_id":"1504.01840","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-for-real-time-atari-game-play","title":"Deep Learning for Real-Time Atari Game Play Using Offline Monte-Carlo Tree Search Planning","date":"2014-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"empirical-q-value-iteration","title":"Empirical Q-Value Iteration","date":"2014-11-30","arxiv_id":"1412.0180","n_code_links":0,"syntology":null},{"paper":null,"slug":"q-learning-for-optimal-control-of-continuous","title":"Q-learning for Optimal Control of Continuous-time Systems","date":"2014-10-11","arxiv_id":"1410.2954","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-algorithm-for","title":"Reinforcement Learning Based Algorithm for the Maximization of EV Charging Station Revenue","date":"2014-07-04","arxiv_id":"1407.1291","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-medical-treatments-using-novel","title":"Personalized Medical Treatments Using Novel Reinforcement Learning Algorithms","date":"2014-06-16","arxiv_id":"1406.3922","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-timescale-convergent-q-learning-for-sleep","title":"Two Timescale Convergent Q-learning for Sleep--Scheduling in Wireless Sensor Networks","date":"2013-12-27","arxiv_id":"1312.7292","n_code_links":0,"syntology":null},{"paper":"/paper/playing-atari-with-deep-reinforcement","slug":"playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","n_code_links":112,"syntology":{"ran":64,"of":117,"n_ran_checked":46,"n_instrument":18,"unverified":53,"pointer_only":56,"phrase":"64 ran (of which 24 constructed an object rather than computing a result; 46 with no instrument failure: 5 honoured, 0 violated, 41 with no contract checked; 18 where Syntology's instrument failed) · 53 unverified","official":null}},{"paper":null,"slug":"q-learning-optimization-in-a-multi-agents","title":"Q-learning optimization in a multi-agents system for image segmentation","date":"2013-11-23","arxiv_id":"1311.6054","n_code_links":0,"syntology":null},{"paper":null,"slug":"risk-sensitive-reinforcement-learning","title":"Risk-sensitive Reinforcement Learning","date":"2013-11-08","arxiv_id":"1311.2097","n_code_links":0,"syntology":null},{"paper":null,"slug":"approximate-kalman-filter-q-learning-for","title":"Approximate Kalman Filter Q-Learning for Continuous State-Space MDPs","date":"2013-09-26","arxiv_id":"1309.6868","n_code_links":0,"syntology":null},{"paper":null,"slug":"projective-simulation-for-classical-learning","title":"Projective simulation for classical learning agents: a comprehensive investigation","date":"2013-05-07","arxiv_id":"1305.1578","n_code_links":0,"syntology":null},{"paper":null,"slug":"speedy-q-learning","title":"Speedy Q-Learning","date":"2011-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/double-q-learning","slug":"double-q-learning","title":"Double Q-learning","date":"2010-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"convergent-temporal-difference-learning-with","title":"Convergent Temporal-Difference Learning with Arbitrary Smooth Function Approximation","date":"2009-12-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"0d8f05103639dee5abdaa49ba8f9a8aad2a828cefb95af278bf2a19284cb4df5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}