{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/prioritized-experience-replay/papers/2","list_of":"/method/prioritized-experience-replay","method":"Prioritized Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,138],"of":138,"counts":{"archive_papers_tagged":138,"with_a_code_link":61,"where_syntology_ran_a_sample":19,"not_listed_spam_title":0,"listed":138,"listed_where_code_ran":19,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":17,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":17,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/prioritized-experience-replay","prev":"/method/prioritized-experience-replay","next":null,"papers":[{"paper":null,"slug":"recurrent-distributed-reinforcement-learning","title":"A Learning Approach to Robot-Agnostic Force-Guided High Precision Assembly","date":"2020-10-15","arxiv_id":"2010.08052","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-approach-for-tactical-decision-making","title":"A New Approach for Tactical Decision Making in Lane Changing: Sample Efficient Deep Q Learning with a Safety Feedback Reward","date":"2020-09-24","arxiv_id":"2009.11905","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-prioritized-replay-sampling-states-in","slug":"beyond-prioritized-replay-sampling-states-in","title":"Understanding and Mitigating the Limitations of Prioritized Experience Replay","date":"2020-07-19","arxiv_id":"2007.09569","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-to-sample-with-local-and-global","title":"Learning to Sample with Local and Global Contexts in Experience Replay Buffer","date":"2020-07-14","arxiv_id":"2007.07358","n_code_links":0,"syntology":null},{"paper":"/paper/sunrise-a-simple-unified-framework-for","slug":"sunrise-a-simple-unified-framework-for","title":"SUNRISE: A Simple Unified Framework for Ensemble Learning in Deep Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04938","n_code_links":1,"syntology":null},{"paper":null,"slug":"double-prioritized-state-recycled-experience","title":"Double Prioritized State Recycled Experience Replay","date":"2020-07-08","arxiv_id":"2007.03961","n_code_links":0,"syntology":null},{"paper":"/paper/the-loca-regret-a-consistent-metric-to","slug":"the-loca-regret-a-consistent-metric-to","title":"The LoCA Regret: A Consistent Metric to Evaluate Model-Based Behavior in Reinforcement Learning","date":"2020-07-07","arxiv_id":"2007.03158","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["chandar-lab/LoCA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"distributed-uplink-beamforming-in-cell-free","title":"Distributed Uplink Beamforming in Cell-Free Networks Using Deep Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.15138","n_code_links":0,"syntology":null},{"paper":null,"slug":"continuous-control-for-searching-and-planning","title":"Continuous Control for Searching and Planning with a Learned Model","date":"2020-06-12","arxiv_id":"2006.07430","n_code_links":0,"syntology":null},{"paper":null,"slug":"balancing-a-cartpole-system-with","title":"Balancing a CartPole System with Reinforcement Learning -- A Tutorial","date":"2020-06-08","arxiv_id":"2006.04938","n_code_links":0,"syntology":null},{"paper":"/paper/manipulating-the-distributions-of-experience","slug":"manipulating-the-distributions-of-experience","title":"Manipulating the Distributions of Experience used for Self-Play Learning in Expert Iteration","date":"2020-05-30","arxiv_id":"2006.00283","n_code_links":1,"syntology":null},{"paper":null,"slug":"stdpg-a-spatio-temporal-deterministic-policy","title":"STDPG: A Spatio-Temporal Deterministic Policy Gradient Agent for Dynamic Routing in SDN","date":"2020-04-21","arxiv_id":"2004.09783","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-experience-replay","title":"Dynamic Experience Replay","date":"2020-03-04","arxiv_id":"2003.02372","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-intelligent","title":"Deep Reinforcement Learning Based Intelligent Reflecting Surface for Secure Wireless Communications","date":"2020-02-27","arxiv_id":"2002.12271","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-reinforcement-learning-for-anti-jamming","title":"Fast Reinforcement Learning for Anti-jamming Communications","date":"2020-02-13","arxiv_id":"2002.05364","n_code_links":0,"syntology":null},{"paper":null,"slug":"stacked-auto-encoder-based-deep-reinforcement","title":"Stacked Auto Encoder Based Deep Reinforcement Learning for Online Resource Scheduling in Large-Scale MEC Networks","date":"2020-01-24","arxiv_id":"2001.09223","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-based-distributional-policy-gradient","title":"Sample-based Distributional Policy Gradient","date":"2020-01-08","arxiv_id":"2001.02652","n_code_links":0,"syntology":null},{"paper":null,"slug":"which-channel-to-ask-my-question-personalized","title":"Which Channel to Ask My Question? Personalized Customer Service RequestStream Routing using DeepReinforcement Learning","date":"2019-11-24","arxiv_id":"1911.10521","n_code_links":0,"syntology":null},{"paper":"/paper/mastering-atari-go-chess-and-shogi-by","slug":"mastering-atari-go-chess-and-shogi-by","title":"Mastering Atari, Go, Chess and Shogi by Planning with a Learned Model","date":"2019-11-19","arxiv_id":"1911.08265","n_code_links":18,"syntology":{"ran":43,"of":64,"n_ran_checked":43,"n_instrument":0,"unverified":21,"pointer_only":62,"phrase":"43 ran (of which 36 constructed an object rather than computing a result; 43 with no instrument failure: 4 honoured, 0 violated, 39 with no contract checked; 0 where Syntology's instrument failed) · 21 unverified","official":null}},{"paper":null,"slug":"placement-optimization-of-aerial-base","title":"Placement Optimization of Aerial Base Stations with Deep Reinforcement Learning","date":"2019-11-19","arxiv_id":"1911.08111","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-dynamic","title":"Deep Reinforcement Learning Based Dynamic Trajectory Control for UAV-assisted Mobile Edge Computing","date":"2019-11-10","arxiv_id":"1911.03887","n_code_links":0,"syntology":null},{"paper":"/paper/task-oriented-language-grounding-for-language","slug":"task-oriented-language-grounding-for-language","title":"Task-Oriented Language Grounding for Language Input with Multiple Sub-Goals of Non-Linear Order","date":"2019-10-27","arxiv_id":"1910.12354","n_code_links":1,"syntology":null},{"paper":"/paper/google-research-football-a-novel","slug":"google-research-football-a-novel","title":"Google Research Football: A Novel Reinforcement Learning Environment","date":"2019-07-25","arxiv_id":"1907.11180","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/football"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"prioritized-guidance-for-efficient-multi","title":"Prioritized Guidance for Efficient Multi-Agent Reinforcement Learning Exploration","date":"2019-07-18","arxiv_id":"1907.07847","n_code_links":0,"syntology":null},{"paper":null,"slug":"190512726","title":"Prioritized Sequence Experience Replay","date":"2019-05-25","arxiv_id":"1905.12726","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-adversarial-imagination-for-sample","title":"Generative Adversarial Imagination for Sample Efficient Deep Reinforcement Learning","date":"2019-04-30","arxiv_id":"1904.13255","n_code_links":0,"syntology":null},{"paper":"/paper/tf-replicator-distributed-machine-learning","slug":"tf-replicator-distributed-machine-learning","title":"TF-Replicator: Distributed Machine Learning for Researchers","date":"2019-02-01","arxiv_id":"1902.00465","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tensorflow/community"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/macro-action-selection-with-deep","slug":"macro-action-selection-with-deep","title":"Macro action selection with deep reinforcement learning in StarCraft","date":"2018-12-02","arxiv_id":"1812.00336","n_code_links":1,"syntology":null},{"paper":"/paper/an-intriguing-failing-of-convolutional-neural","slug":"an-intriguing-failing-of-convolutional-neural","title":"An Intriguing Failing of Convolutional Neural Networks and the CoordConv Solution","date":"2018-07-09","arxiv_id":"1807.03247","n_code_links":24,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uber-research/coordconv"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"deep-curiosity-search-intra-life-exploration","title":"Deep Curiosity Search: Intra-Life Exploration Can Improve Performance on Challenging Deep Reinforcement Learning Problems","date":"2018-06-01","arxiv_id":"1806.00553","n_code_links":0,"syntology":null},{"paper":"/paper/advances-in-experience-replay","slug":"advances-in-experience-replay","title":"Advances in Experience Replay","date":"2018-05-15","arxiv_id":"1805.05536","n_code_links":1,"syntology":null},{"paper":"/paper/distributed-distributional-deterministic","slug":"distributed-distributional-deterministic","title":"Distributed Distributional Deterministic Policy Gradients","date":"2018-04-23","arxiv_id":"1804.08617","n_code_links":5,"syntology":null},{"paper":"/paper/distributed-prioritized-experience-replay","slug":"distributed-prioritized-experience-replay","title":"Distributed Prioritized Experience Replay","date":"2018-03-02","arxiv_id":"1803.00933","n_code_links":15,"syntology":{"ran":9,"of":15,"n_ran_checked":9,"n_instrument":0,"unverified":6,"pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":null,"slug":"screenernet-learning-self-paced-curriculum","title":"ScreenerNet: Learning Self-Paced Curriculum for Deep Neural Networks","date":"2018-01-03","arxiv_id":"1801.00904","n_code_links":0,"syntology":null},{"paper":null,"slug":"vizdoom-drqn-with-prioritized-experience","title":"ViZDoom: DRQN with Prioritized Experience Replay, Double-Q Learning, & Snapshot Ensembling","date":"2018-01-03","arxiv_id":"1801.01000","n_code_links":0,"syntology":null},{"paper":"/paper/rainbow-combining-improvements-in-deep","slug":"rainbow-combining-improvements-in-deep","title":"Rainbow: Combining Improvements in Deep Reinforcement Learning","date":"2017-10-06","arxiv_id":"1710.02298","n_code_links":34,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/a-novel-ddpg-method-with-prioritized","slug":"a-novel-ddpg-method-with-prioritized","title":"A novel DDPG method with prioritized experience replay","date":"2017-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/prioritized-experience-replay","slug":"prioritized-experience-replay","title":"Prioritized Experience Replay","date":"2015-11-18","arxiv_id":"1511.05952","n_code_links":77,"syntology":{"ran":80,"of":111,"n_ran_checked":72,"n_instrument":8,"unverified":31,"pointer_only":43,"phrase":"80 ran (of which 62 constructed an object rather than computing a result; 72 with no instrument failure: 4 honoured, 0 violated, 68 with no contract checked; 8 where Syntology's instrument failed) · 31 unverified","official":null}}],"record_sha256":"61f07733d3dc1368d0d2b11ae74ce60316b987ef50e17bc84f7c6a1b72083fcd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}