{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/double-q-learning/papers/2","list_of":"/method/double-q-learning","method":"Double Q-learning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,112],"of":112,"counts":{"archive_papers_tagged":112,"with_a_code_link":45,"where_syntology_ran_a_sample":15,"not_listed_spam_title":0,"listed":112,"listed_where_code_ran":15,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":13,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":13,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/double-q-learning","prev":"/method/double-q-learning","next":null,"papers":[{"paper":"/paper/distributed-prioritized-experience-replay","slug":"distributed-prioritized-experience-replay","title":"Distributed Prioritized Experience Replay","date":"2018-03-02","arxiv_id":"1803.00933","n_code_links":15,"syntology":{"ran":9,"of":15,"n_ran_checked":9,"n_instrument":0,"unverified":6,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/addressing-function-approximation-error-in","slug":"addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","n_code_links":67,"syntology":{"ran":26,"of":36,"n_ran_checked":25,"n_instrument":1,"unverified":10,"pointer_only":21,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 25 with no instrument failure: 1 honoured, 1 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 10 unverified","official":{"repos":["sfujim/TD3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"weighted-double-deep-multiagent-reinforcement","title":"Weighted Double Deep Multiagent Reinforcement Learning in Stochastic Cooperative Environments","date":"2018-02-23","arxiv_id":"1802.08534","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-exploration-through-bayesian-deep-q","slug":"efficient-exploration-through-bayesian-deep-q","title":"Efficient Exploration through Bayesian Deep Q-Networks","date":"2018-02-13","arxiv_id":"1802.04412","n_code_links":1,"syntology":null},{"paper":null,"slug":"faster-deep-q-learning-using-neural-episodic","title":"Faster Deep Q-learning using Neural Episodic Control","date":"2018-01-06","arxiv_id":"1801.01968","n_code_links":0,"syntology":null},{"paper":"/paper/rainbow-combining-improvements-in-deep","slug":"rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","n_code_links":34,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/noisy-networks-for-exploration","slug":"noisy-networks-for-exploration","title":"Noisy Networks for Exploration","date":"2017-06-30","arxiv_id":"1706.10295","n_code_links":15,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/sample-efficient-actor-critic-with-experience","slug":"sample-efficient-actor-critic-with-experience","title":"Sample Efficient Actor-Critic with Experience Replay","date":"2016-11-03","arxiv_id":"1611.01224","n_code_links":7,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":null}},{"paper":null,"slug":"dynamic-frame-skip-deep-q-network","title":"Dynamic Frame skip Deep Q Network","date":"2016-05-17","arxiv_id":"1605.05365","n_code_links":0,"syntology":null},{"paper":"/paper/dueling-network-architectures-for-deep","slug":"dueling-network-architectures-for-deep","title":"Dueling Network Architectures for Deep Reinforcement Learning","date":"2015-11-20","arxiv_id":"1511.06581","n_code_links":73,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/deep-reinforcement-learning-with-double-q","slug":"deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","n_code_links":97,"syntology":{"ran":56,"of":106,"n_ran_checked":55,"n_instrument":1,"unverified":50,"pointer_only":57,"phrase":"56 ran (of which 38 constructed an object rather than computing a result; 55 with no instrument failure: 0 honoured, 0 violated, 55 with no contract checked; 1 where Syntology's instrument failed) · 50 unverified","official":null}},{"paper":"/paper/double-q-learning","slug":"double-q-learning","title":"Double Q-learning","date":"2010-12-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"93a49813ee8fc7658ff200cf6e2c9739ca9c1de5e55c81ecc2e57ccfedb7c0e8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}