{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/dqn/papers/6","list_of":"/method/dqn","method":"DQN","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":6,"rows_per_page":100,"rows":[501,519],"of":519,"counts":{"archive_papers_tagged":519,"with_a_code_link":173,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":519,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":36,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":36,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/dqn","prev":"/method/dqn/papers/5","next":null,"papers":[{"paper":"/paper/deep-reinforcement-learning-for-multi-domain","slug":"deep-reinforcement-learning-for-multi-domain","title":"Deep Reinforcement Learning for Multi-Domain Dialogue Systems","date":"2016-11-26","arxiv_id":"1611.08675","n_code_links":1,"syntology":null},{"paper":null,"slug":"memory-lens-how-much-memory-does-an-agent-use","title":"Memory Lens: How Much Memory Does an Agent Use?","date":"2016-11-21","arxiv_id":"1611.06928","n_code_links":0,"syntology":null},{"paper":null,"slug":"averaged-dqn-variance-reduction-and","title":"Averaged-DQN: Variance Reduction and Stabilization for Deep Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.01929","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-from-raw-pixels","title":"Deep Reinforcement Learning From Raw Pixels in Doom","date":"2016-10-07","arxiv_id":"1610.02164","n_code_links":0,"syntology":null},{"paper":"/paper/opponent-modeling-in-deep-reinforcement","slug":"opponent-modeling-in-deep-reinforcement","title":"Opponent Modeling in Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05559","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-discovers","title":"Deep Reinforcement Learning Discovers Internal Models","date":"2016-06-16","arxiv_id":"1606.05174","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-macro","title":"Deep Reinforcement Learning With Macro-Actions","date":"2016-06-15","arxiv_id":"1606.04615","n_code_links":0,"syntology":null},{"paper":null,"slug":"classifying-options-for-deep-reinforcement","title":"Classifying Options for Deep Reinforcement Learning","date":"2016-04-27","arxiv_id":"1604.08153","n_code_links":0,"syntology":null},{"paper":"/paper/deep-exploration-via-bootstrapped-dqn","slug":"deep-exploration-via-bootstrapped-dqn","title":"Deep Exploration via Bootstrapped DQN","date":"2016-02-15","arxiv_id":"1602.04621","n_code_links":6,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"how-to-discount-deep-reinforcement-learning","title":"How to Discount Deep Reinforcement Learning: Towards New Dynamic Strategies","date":"2015-12-07","arxiv_id":"1512.02011","n_code_links":0,"syntology":null},{"paper":"/paper/deep-attention-recurrent-q-network","slug":"deep-attention-recurrent-q-network","title":"Deep Attention Recurrent Q-Network","date":"2015-12-05","arxiv_id":"1512.01693","n_code_links":3,"syntology":null},{"paper":"/paper/state-of-the-art-control-of-atari-games-using","slug":"state-of-the-art-control-of-atari-games-using","title":"State of the Art Control of Atari Games Using Shallow Reinforcement Learning","date":"2015-12-04","arxiv_id":"1512.01563","n_code_links":1,"syntology":null},{"paper":"/paper/policy-distillation","slug":"policy-distillation","title":"Policy Distillation","date":"2015-11-19","arxiv_id":"1511.06295","n_code_links":1,"syntology":null},{"paper":"/paper/prioritized-experience-replay","slug":"prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","arxiv_id":"1511.05952","n_code_links":77,"syntology":{"ran":80,"of":111,"n_ran_checked":72,"n_instrument":8,"unverified":31,"pointer_only":43,"phrase":"80 ran (of which 62 constructed an object rather than computing a result; 72 with no instrument failure: 4 honoured, 0 violated, 68 with no contract checked; 8 where Syntology's instrument failed) · 31 unverified","official":null}},{"paper":null,"slug":"generating-text-with-deep-reinforcement","title":"Generating Text with Deep Reinforcement Learning","date":"2015-10-30","arxiv_id":"1510.09202","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-double-q","slug":"deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","n_code_links":97,"syntology":{"ran":56,"of":106,"n_ran_checked":55,"n_instrument":1,"unverified":50,"pointer_only":57,"phrase":"56 ran (of which 38 constructed an object rather than computing a result; 55 with no instrument failure: 0 honoured, 0 violated, 55 with no contract checked; 1 where Syntology's instrument failed) · 50 unverified","official":null}},{"paper":"/paper/massively-parallel-methods-for-deep","slug":"massively-parallel-methods-for-deep","title":"Massively Parallel Methods for Deep Reinforcement Learning","date":"2015-07-15","arxiv_id":"1507.04296","n_code_links":3,"syntology":null},{"paper":null,"slug":"deep-learning-for-real-time-atari-game-play","title":"Deep Learning for Real-Time Atari Game Play Using Offline Monte-Carlo Tree Search Planning","date":"2014-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/playing-atari-with-deep-reinforcement","slug":"playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","n_code_links":112,"syntology":{"ran":64,"of":117,"n_ran_checked":46,"n_instrument":18,"unverified":53,"pointer_only":56,"phrase":"64 ran (of which 24 constructed an object rather than computing a result; 46 with no instrument failure: 5 honoured, 0 violated, 41 with no contract checked; 18 where Syntology's instrument failed) · 53 unverified","official":null}}],"record_sha256":"af12105b8a16c3f58b8f9e16d2c260783293b46d1c89a938d3f7c400cd26a554","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}