{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/9","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":9,"rows_per_page":100,"rows":[801,865],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay/papers/8","next":null,"papers":[{"paper":"/paper/dynamic-weights-in-multi-objective-deep","slug":"dynamic-weights-in-multi-objective-deep","title":"Dynamic Weights in Multi-Objective Deep Reinforcement Learning","date":"2018-09-20","arxiv_id":"1809.07803","n_code_links":3,"syntology":null},{"paper":null,"slug":"curriculum-goal-masking-for-continuous-deep","title":"Curriculum goal masking for continuous deep reinforcement learning","date":"2018-09-17","arxiv_id":"1809.06146","n_code_links":0,"syntology":null},{"paper":"/paper/generalizing-across-multi-objective-reward","slug":"generalizing-across-multi-objective-reward","title":"Generalizing Across Multi-Objective Reward Functions in Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06364","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"improvements-on-hindsight-learning","title":"Improvements on Hindsight Learning","date":"2018-09-16","arxiv_id":"1809.06719","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-advertise-with-adaptive-exposure","title":"Learning Adaptive Display Exposure for Real-Time Advertising","date":"2018-09-10","arxiv_id":"1809.03149","n_code_links":0,"syntology":null},{"paper":"/paper/archer-aggressive-rewards-to-counter-bias-in","slug":"archer-aggressive-rewards-to-counter-bias-in","title":"ARCHER: Aggressive Rewards to Counter bias in Hindsight Experience Replay","date":"2018-09-06","arxiv_id":"1809.02070","n_code_links":1,"syntology":null},{"paper":"/paper/adversarial-deep-reinforcement-learning-in","slug":"adversarial-deep-reinforcement-learning-in","title":"Adversarial Deep Reinforcement Learning in Portfolio Management","date":"2018-08-29","arxiv_id":"1808.09940","n_code_links":5,"syntology":null},{"paper":null,"slug":"goal-oriented-dialogue-policy-learning-from","title":"Goal-oriented Dialogue Policy Learning from Failures","date":"2018-08-20","arxiv_id":"1808.06497","n_code_links":0,"syntology":null},{"paper":"/paper/bipedal-walking-robot-using-deep","slug":"bipedal-walking-robot-using-deep","title":"Bipedal Walking Robot using Deep Deterministic Policy Gradient","date":"2018-07-16","arxiv_id":"1807.05924","n_code_links":3,"syntology":null},{"paper":"/paper/remember-and-forget-for-experience-replay","slug":"remember-and-forget-for-experience-replay","title":"Remember and Forget for Experience Replay","date":"2018-07-16","arxiv_id":"1807.05827","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cselab/smarties"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deterministic-policy-gradients-with-general","title":"Deterministic Policy Gradients With General State Transitions","date":"2018-07-10","arxiv_id":"1807.03708","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-explore-via-meta-policy-gradient","title":"Learning to Explore via Meta-Policy Gradient","date":"2018-07-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/stroke-based-character-reconstruction","slug":"stroke-based-character-reconstruction","title":"Stroke-based Character Reconstruction","date":"2018-06-23","arxiv_id":"1806.08990","n_code_links":1,"syntology":null},{"paper":null,"slug":"organizing-experience-a-deeper-look-at-replay","title":"Organizing Experience: A Deeper Look at Replay Mechanisms for Sample-based Planning in Continuous State Domains","date":"2018-06-12","arxiv_id":"1806.04624","n_code_links":0,"syntology":null},{"paper":"/paper/randomized-value-functions-via-multiplicative","slug":"randomized-value-functions-via-multiplicative","title":"Randomized Value Functions via Multiplicative Normalizing Flows","date":"2018-06-06","arxiv_id":"1806.02315","n_code_links":2,"syntology":null},{"paper":"/paper/sample-efficient-deep-reinforcement-learning-2","slug":"sample-efficient-deep-reinforcement-learning-2","title":"Sample-Efficient Deep Reinforcement Learning via Episodic Backward Update","date":"2018-05-31","arxiv_id":"1805.12375","n_code_links":1,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-in-a-handful-of","slug":"deep-reinforcement-learning-in-a-handful-of","title":"Deep Reinforcement Learning in a Handful of Trials using Probabilistic Dynamics Models","date":"2018-05-30","arxiv_id":"1805.12114","n_code_links":9,"syntology":{"ran":12,"of":19,"n_ran_checked":7,"n_instrument":5,"unverified":7,"pointer_only":17,"phrase":"12 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 7 unverified","official":{"repos":["kchua/handful-of-trials"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/advances-in-experience-replay","slug":"advances-in-experience-replay","title":"Advances in Experience Replay","date":"2018-05-15","arxiv_id":"1805.05536","n_code_links":1,"syntology":null},{"paper":"/paper/do-deep-reinforcement-learning-agents-model","slug":"do-deep-reinforcement-learning-agents-model","title":"Do deep reinforcement learning agents model intentions?","date":"2018-05-15","arxiv_id":"1805.06020","n_code_links":1,"syntology":null},{"paper":null,"slug":"metatrace-online-step-size-tuning-by-meta","title":"Metatrace Actor-Critic: Online Step-size Tuning by Meta-gradient Descent for Reinforcement Learning Control","date":"2018-05-10","arxiv_id":"1805.04514","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiagent-soft-q-learning","title":"Multiagent Soft Q-Learning","date":"2018-04-25","arxiv_id":"1804.09817","n_code_links":0,"syntology":null},{"paper":null,"slug":"state-distribution-aware-sampling-for-deep-q","title":"State Distribution-aware Sampling for Deep Q-learning","date":"2018-04-23","arxiv_id":"1804.08619","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-explore-with-meta-policy-gradient","title":"Learning to Explore with Meta-Policy Gradient","date":"2018-03-13","arxiv_id":"1803.05044","n_code_links":0,"syntology":null},{"paper":"/paper/distributed-prioritized-experience-replay","slug":"distributed-prioritized-experience-replay","title":"Distributed Prioritized Experience Replay","date":"2018-03-02","arxiv_id":"1803.00933","n_code_links":15,"syntology":{"ran":9,"of":15,"n_ran_checked":9,"n_instrument":0,"unverified":6,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/addressing-function-approximation-error-in","slug":"addressing-function-approximation-error-in","title":"Addressing Function Approximation Error in Actor-Critic Methods","date":"2018-02-26","arxiv_id":"1802.09477","n_code_links":67,"syntology":{"ran":26,"of":36,"n_ran_checked":25,"n_instrument":1,"unverified":10,"pointer_only":21,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 25 with no instrument failure: 1 honoured, 1 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 10 unverified","official":{"repos":["sfujim/TD3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/multi-goal-reinforcement-learning-challenging","slug":"multi-goal-reinforcement-learning-challenging","title":"Multi-Goal Reinforcement Learning: Challenging Robotics Environments and Request for Research","date":"2018-02-26","arxiv_id":"1802.09464","n_code_links":28,"syntology":null},{"paper":null,"slug":"weighted-double-deep-multiagent-reinforcement","title":"Weighted Double Deep Multiagent Reinforcement Learning in Stochastic Cooperative Environments","date":"2018-02-23","arxiv_id":"1802.08534","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-reinforcement-learning-with-complex","title":"Continual Reinforcement Learning with Complex Synapses","date":"2018-02-20","arxiv_id":"1802.07239","n_code_links":0,"syntology":null},{"paper":"/paper/a-deep-q-learning-agent-for-the-l-game-with","slug":"a-deep-q-learning-agent-for-the-l-game-with","title":"A Deep Q-Learning Agent for the L-Game with Variable Batch Training","date":"2018-02-17","arxiv_id":"1802.06225","n_code_links":1,"syntology":null},{"paper":"/paper/gep-pg-decoupling-exploration-and","slug":"gep-pg-decoupling-exploration-and","title":"GEP-PG: Decoupling Exploration and Exploitation in Deep Reinforcement Learning Algorithms","date":"2018-02-14","arxiv_id":"1802.05054","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["flowersteam/geppg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/efficient-exploration-through-bayesian-deep-q","slug":"efficient-exploration-through-bayesian-deep-q","title":"Efficient Exploration through Bayesian Deep Q-Networks","date":"2018-02-13","arxiv_id":"1802.04412","n_code_links":1,"syntology":null},{"paper":null,"slug":"sample-efficient-deep-reinforcement-learning","title":"Sample Efficient Deep Reinforcement Learning for Dialogue Systems with Large Action Spaces","date":"2018-02-11","arxiv_id":"1802.03753","n_code_links":0,"syntology":null},{"paper":"/paper/impala-scalable-distributed-deep-rl-with","slug":"impala-scalable-distributed-deep-rl-with","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","date":"2018-02-05","arxiv_id":"1802.01561","n_code_links":24,"syntology":{"ran":16,"of":34,"n_ran_checked":10,"n_instrument":6,"unverified":18,"pointer_only":3,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 18 unverified","official":{"repos":["deepmind/scalable_agent"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"pretraining-deep-actor-critic-reinforcement","title":"Pretraining Deep Actor-Critic Reinforcement Learning Algorithms With Expert Demonstrations","date":"2018-01-31","arxiv_id":"1801.10459","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-in-gpu-experience-replay","title":"Deep In-GPU Experience Replay","date":"2018-01-09","arxiv_id":"1801.03138","n_code_links":0,"syntology":null},{"paper":null,"slug":"faster-deep-q-learning-using-neural-episodic","title":"Faster Deep Q-learning using Neural Episodic Control","date":"2018-01-06","arxiv_id":"1801.01968","n_code_links":0,"syntology":null},{"paper":"/paper/soft-actor-critic-off-policy-maximum-entropy","slug":"soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","arxiv_id":"1801.01290","n_code_links":86,"syntology":{"ran":91,"of":148,"n_ran_checked":73,"n_instrument":18,"unverified":57,"pointer_only":66,"phrase":"91 ran (of which 61 constructed an object rather than computing a result; 73 with no instrument failure: 3 honoured, 1 violated, 69 with no contract checked; 18 where Syntology's instrument failed) · 57 unverified","official":{"repos":["haarnoja/sac"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"screenernet-learning-self-paced-curriculum","title":"ScreenerNet: Learning Self-Paced Curriculum for Deep Neural Networks","date":"2018-01-03","arxiv_id":"1801.00904","n_code_links":0,"syntology":null},{"paper":null,"slug":"vizdoom-drqn-with-prioritized-experience","title":"ViZDoom: DRQN with Prioritized Experience Replay, Double-Q Learning, & Snapshot Ensembling","date":"2018-01-03","arxiv_id":"1801.01000","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-run-with-actor-critic-ensemble","slug":"learning-to-run-with-actor-critic-ensemble","title":"Learning to Run with Actor-Critic Ensemble","date":"2017-12-25","arxiv_id":"1712.08987","n_code_links":2,"syntology":null},{"paper":"/paper/a-deeper-look-at-experience-replay","slug":"a-deeper-look-at-experience-replay","title":"A Deeper Look at Experience Replay","date":"2017-12-04","arxiv_id":"1712.01275","n_code_links":4,"syntology":null},{"paper":null,"slug":"amber-adaptive-multi-batch-experience-replay","title":"AMBER: Adaptive Multi-Batch Experience Replay for Continuous Action Control","date":"2017-10-12","arxiv_id":"1710.04423","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-ddpg-method-with-prioritized","slug":"a-novel-ddpg-method-with-prioritized","title":"A novel DDPG method with prioritized experience replay","date":"2017-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/overcoming-exploration-in-reinforcement","slug":"overcoming-exploration-in-reinforcement","title":"Overcoming Exploration in Reinforcement Learning with Demonstrations","date":"2017-09-28","arxiv_id":"1709.10089","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"linear-stochastic-approximation-constant-step","title":"Linear Stochastic Approximation: Constant Step-Size and Iterate Averaging","date":"2017-09-12","arxiv_id":"1709.04073","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-stochastic-policy-gradients-in","title":"Improving Stochastic Policy Gradients in Continuous Control with Deep Reinforcement Learning using the Beta Distribution","date":"2017-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-demonstrations-for-deep","slug":"leveraging-demonstrations-for-deep","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","date":"2017-07-27","arxiv_id":"1707.08817","n_code_links":4,"syntology":null},{"paper":"/paper/lenient-multi-agent-deep-reinforcement","slug":"lenient-multi-agent-deep-reinforcement","title":"Lenient Multi-Agent Deep Reinforcement Learning","date":"2017-07-14","arxiv_id":"1707.04402","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-intentional-unintentional-agent-learning","title":"The Intentional Unintentional Agent: Learning to Solve Many Continuous Control Tasks Simultaneously","date":"2017-07-11","arxiv_id":"1707.03300","n_code_links":0,"syntology":null},{"paper":"/paper/hindsight-experience-replay","slug":"hindsight-experience-replay","title":"Hindsight Experience Replay","date":"2017-07-05","arxiv_id":"1707.01495","n_code_links":28,"syntology":{"ran":16,"of":16,"n_ran_checked":11,"n_instrument":5,"unverified":0,"pointer_only":9,"phrase":"16 ran (of which 10 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"sample-efficient-actor-critic-reinforcement","title":"Sample-efficient Actor-Critic Reinforcement Learning with Supervised Data for Dialogue Management","date":"2017-07-01","arxiv_id":"1707.00130","n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-actor-critic-for-mixed","slug":"multi-agent-actor-critic-for-mixed","title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","date":"2017-06-07","arxiv_id":"1706.02275","n_code_links":86,"syntology":{"ran":75,"of":143,"n_ran_checked":68,"n_instrument":7,"unverified":68,"pointer_only":99,"phrase":"75 ran (of which 54 constructed an object rather than computing a result; 68 with no instrument failure: 2 honoured, 0 violated, 66 with no contract checked; 7 where Syntology's instrument failed) · 68 unverified","official":{"repos":["openai/multiagent-particle-envs"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/parameter-space-noise-for-exploration","slug":"parameter-space-noise-for-exploration","title":"Parameter Space Noise for Exploration","date":"2017-06-06","arxiv_id":"1706.01905","n_code_links":10,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/continual-learning-with-deep-generative","slug":"continual-learning-with-deep-generative","title":"Continual Learning with Deep Generative Replay","date":"2017-05-24","arxiv_id":"1705.08690","n_code_links":5,"syntology":null},{"paper":null,"slug":"discrete-sequential-prediction-of-continuous","title":"Discrete Sequential Prediction of Continuous Actions for Deep RL","date":"2017-05-14","arxiv_id":"1705.05035","n_code_links":0,"syntology":null},{"paper":"/paper/stabilising-experience-replay-for-deep-multi","slug":"stabilising-experience-replay-for-deep-multi","title":"Stabilising Experience Replay for Deep Multi-Agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08887","n_code_links":5,"syntology":null},{"paper":null,"slug":"sample-efficient-deep-reinforcement-learning-1","title":"Sample-efficient Deep Reinforcement Learning for Dialog Control","date":"2016-12-18","arxiv_id":"1612.06000","n_code_links":0,"syntology":null},{"paper":"/paper/sample-efficient-actor-critic-with-experience","slug":"sample-efficient-actor-critic-with-experience","title":"Sample Efficient Actor-Critic with Experience Replay","date":"2016-11-03","arxiv_id":"1611.01224","n_code_links":7,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":null}},{"paper":null,"slug":"online-contrastive-divergence-with-generative","title":"Online Contrastive Divergence with Generative Replay: Experience Replay without Storing Data","date":"2016-10-18","arxiv_id":"1610.05555","n_code_links":0,"syntology":null},{"paper":"/paper/actor-critic-versus-direct-policy-search-a","slug":"actor-critic-versus-direct-policy-search-a","title":"Actor-critic versus direct policy search: a comparison based on sample complexity","date":"2016-06-29","arxiv_id":"1606.09152","n_code_links":1,"syntology":null},{"paper":"/paper/continuous-deep-q-learning-with-model-based","slug":"continuous-deep-q-learning-with-model-based","title":"Continuous Deep Q-Learning with Model-based Acceleration","date":"2016-03-02","arxiv_id":"1603.00748","n_code_links":8,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/prioritized-experience-replay","slug":"prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","arxiv_id":"1511.05952","n_code_links":77,"syntology":{"ran":80,"of":111,"n_ran_checked":72,"n_instrument":8,"unverified":31,"pointer_only":43,"phrase":"80 ran (of which 62 constructed an object rather than computing a result; 72 with no instrument failure: 4 honoured, 0 violated, 68 with no contract checked; 8 where Syntology's instrument failed) · 31 unverified","official":null}},{"paper":"/paper/deep-reinforcement-learning-with-double-q","slug":"deep-reinforcement-learning-with-double-q","title":"Deep Reinforcement Learning with Double Q-learning","date":"2015-09-22","arxiv_id":"1509.06461","n_code_links":97,"syntology":{"ran":56,"of":106,"n_ran_checked":55,"n_instrument":1,"unverified":50,"pointer_only":57,"phrase":"56 ran (of which 38 constructed an object rather than computing a result; 55 with no instrument failure: 0 honoured, 0 violated, 55 with no contract checked; 1 where Syntology's instrument failed) · 50 unverified","official":null}},{"paper":"/paper/continuous-control-with-deep-reinforcement","slug":"continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","arxiv_id":"1509.02971","n_code_links":161,"syntology":{"ran":163,"of":306,"n_ran_checked":152,"n_instrument":11,"unverified":143,"pointer_only":163,"phrase":"163 ran (of which 126 constructed an object rather than computing a result; 152 with no instrument failure: 3 honoured, 0 violated, 149 with no contract checked; 11 where Syntology's instrument failed) · 143 unverified","official":null}},{"paper":"/paper/playing-atari-with-deep-reinforcement","slug":"playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","n_code_links":112,"syntology":{"ran":64,"of":117,"n_ran_checked":46,"n_instrument":18,"unverified":53,"pointer_only":56,"phrase":"64 ran (of which 24 constructed an object rather than computing a result; 46 with no instrument failure: 5 honoured, 0 violated, 41 with no contract checked; 18 where Syntology's instrument failed) · 53 unverified","official":null}}],"record_sha256":"9e95ea1fde88672d4fb20b2be0cb7c67e442ec67cd074643ef6de834d92c1ad9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}