{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/7","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":9,"rows_per_page":100,"rows":[601,700],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay/papers/6","next":"/method/experience-replay/papers/8","papers":[{"paper":"/paper/student-initiated-action-advising-via-advice","slug":"student-initiated-action-advising-via-advice","title":"Student-Initiated Action Advising via Advice Novelty","date":"2020-10-01","arxiv_id":"2010.00381","n_code_links":1,"syntology":null},{"paper":"/paper/lucid-dreaming-for-experience-replay","slug":"lucid-dreaming-for-experience-replay","title":"Lucid Dreaming for Experience Replay: Refreshing Past States with the Current Policy","date":"2020-09-29","arxiv_id":"2009.13736","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-assisted-deep-reinforcement","title":"Knowledge-Assisted Deep Reinforcement Learning in 5G Scheduler Design: From Theoretical Framework to Implementation","date":"2020-09-17","arxiv_id":"2009.08346","n_code_links":0,"syntology":null},{"paper":"/paper/meta-learning-with-sparse-experience-replay","slug":"meta-learning-with-sparse-experience-replay","title":"Meta-Learning with Sparse Experience Replay for Lifelong Language Learning","date":"2020-09-10","arxiv_id":"2009.04891","n_code_links":1,"syntology":null},{"paper":"/paper/sample-efficient-automated-deep-reinforcement","slug":"sample-efficient-automated-deep-reinforcement","title":"Sample-Efficient Automated Deep Reinforcement Learning","date":"2020-09-03","arxiv_id":"2009.01555","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["automl/SEARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/adversarial-shapley-value-experience-replay","slug":"adversarial-shapley-value-experience-replay","title":"Online Class-Incremental Continual Learning with Adversarial Shapley Value","date":"2020-08-31","arxiv_id":"2009.00093","n_code_links":3,"syntology":null},{"paper":null,"slug":"sample-efficiency-in-sparse-reinforcement","title":"Sample Efficiency in Sparse Reinforcement Learning: Or Your Money Back","date":"2020-08-28","arxiv_id":"2008.12693","n_code_links":0,"syntology":null},{"paper":null,"slug":"curriculum-learning-with-hindsight-experience","title":"Curriculum Learning with Hindsight Experience Replay for Sequential Object Manipulation Tasks","date":"2020-08-21","arxiv_id":"2008.09377","n_code_links":0,"syntology":null},{"paper":null,"slug":"chrome-dino-run-using-reinforcement-learning","title":"Chrome Dino Run using Reinforcement Learning","date":"2020-08-15","arxiv_id":"2008.06799","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-quantum","slug":"reinforcement-learning-with-quantum","title":"Reinforcement Learning with Quantum Variational Circuits","date":"2020-08-15","arxiv_id":"2008.07524","n_code_links":2,"syntology":null},{"paper":null,"slug":"hierarchial-reinforcement-learning-in","title":"Hierarchical Reinforcement Learning in StarCraft II with Human Expertise in Subgoals Selection","date":"2020-08-08","arxiv_id":"2008.03444","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-q-network-based-multi-agent","title":"Deep Q-Network Based Multi-agent Reinforcement Learning with Binary Action Agents","date":"2020-08-06","arxiv_id":"2008.04109","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimization-driven-hierarchical-learning","title":"Optimization-driven Hierarchical Learning Framework for Wireless Powered Backscatter-aided Relay Communications","date":"2020-08-04","arxiv_id":"2008.01366","n_code_links":0,"syntology":null},{"paper":"/paper/complex-robotic-manipulation-via-graph-based","slug":"complex-robotic-manipulation-via-graph-based","title":"Complex Robotic Manipulation via Graph-Based Hindsight Goal Generation","date":"2020-07-27","arxiv_id":"2007.13486","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-compositional-neural-programs-for","title":"Learning Compositional Neural Programs for Continuous Control","date":"2020-07-27","arxiv_id":"2007.13363","n_code_links":0,"syntology":null},{"paper":null,"slug":"emaq-expected-max-q-learning-operator-for","title":"EMaQ: Expected-Max Q-Learning Operator for Simple Yet Effective Offline and Online RL","date":"2020-07-21","arxiv_id":"2007.11091","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-prioritized-replay-sampling-states-in","slug":"beyond-prioritized-replay-sampling-states-in","title":"Understanding and Mitigating the Limitations of Prioritized Experience Replay","date":"2020-07-19","arxiv_id":"2007.09569","n_code_links":2,"syntology":null},{"paper":"/paper/collision-avoidance-robotics-via-meta","slug":"collision-avoidance-robotics-via-meta","title":"Collision Avoidance Robotics Via Meta-Learning (CARML)","date":"2020-07-16","arxiv_id":"2007.08616","n_code_links":1,"syntology":null},{"paper":null,"slug":"human-like-energy-management-based-on-deep","title":"Human-like Energy Management Based on Deep Reinforcement Learning and Historical Driving Experiences","date":"2020-07-16","arxiv_id":"2007.10126","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-sample-with-local-and-global","title":"Learning to Sample with Local and Global Contexts in Experience Replay Buffer","date":"2020-07-14","arxiv_id":"2007.07358","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-fundamentals-of-experience-replay","slug":"revisiting-fundamentals-of-experience-replay","title":"Revisiting Fundamentals of Experience Replay","date":"2020-07-13","arxiv_id":"2007.06700","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/an-equivalence-between-loss-functions-and-non","slug":"an-equivalence-between-loss-functions-and-non","title":"An Equivalence between Loss Functions and Non-Uniform Sampling in Experience Replay","date":"2020-07-12","arxiv_id":"2007.06049","n_code_links":2,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["sfujim/LAP-PAL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/batch-level-experience-replay-with-review-for","slug":"batch-level-experience-replay-with-review-for","title":"Batch-level Experience Replay with Review for Continual Learning","date":"2020-07-11","arxiv_id":"2007.05683","n_code_links":1,"syntology":null},{"paper":null,"slug":"double-prioritized-state-recycled-experience","title":"Double Prioritized State Recycled Experience Replay","date":"2020-07-08","arxiv_id":"2007.03961","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-policy-adaptation-during","slug":"self-supervised-policy-adaptation-during","title":"Self-Supervised Policy Adaptation during Deployment","date":"2020-07-08","arxiv_id":"2007.04309","n_code_links":2,"syntology":{"ran":16,"of":21,"n_ran_checked":10,"n_instrument":6,"unverified":5,"pointer_only":21,"phrase":"16 ran (of which 9 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/counterfactual-data-augmentation-using","slug":"counterfactual-data-augmentation-using","title":"Counterfactual Data Augmentation using Locally Factored Dynamics","date":"2020-07-06","arxiv_id":"2007.02863","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"developing-cooperative-policies-for-multi","title":"Developing cooperative policies for multi-stage tasks","date":"2020-07-01","arxiv_id":"2007.00203","n_code_links":0,"syntology":null},{"paper":null,"slug":"regularly-updated-deterministic-policy","title":"Regularly Updated Deterministic Policy Gradient Algorithm","date":"2020-07-01","arxiv_id":"2007.00169","n_code_links":0,"syntology":null},{"paper":"/paper/uav-path-planning-for-wireless-data","slug":"uav-path-planning-for-wireless-data","title":"UAV Path Planning for Wireless Data Harvesting: A Deep Reinforcement Learning Approach","date":"2020-07-01","arxiv_id":"2007.00544","n_code_links":3,"syntology":null},{"paper":null,"slug":"distributed-uplink-beamforming-in-cell-free","title":"Distributed Uplink Beamforming in Cell-Free Networks Using Deep Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.15138","n_code_links":0,"syntology":null},{"paper":null,"slug":"noise-overestimation-and-exploration-in-deep","title":"Some approaches used to overcome overestimation in Deep Reinforcement Learning algorithms","date":"2020-06-25","arxiv_id":"2006.14167","n_code_links":0,"syntology":null},{"paper":"/paper/experience-replay-with-likelihood-free","slug":"experience-replay-with-likelihood-free","title":"Experience Replay with Likelihood-free Importance Weights","date":"2020-06-23","arxiv_id":"2006.13169","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"the-effect-of-multi-step-methods-on","title":"The Effect of Multi-step Methods on Overestimation in Deep Reinforcement Learning","date":"2020-06-23","arxiv_id":"2006.12692","n_code_links":0,"syntology":null},{"paper":null,"slug":"autood-automated-outlier-detection-via","title":"AutoOD: Automated Outlier Detection via Curiosity-guided Search and Self-imitation Learning","date":"2020-06-19","arxiv_id":"2006.11321","n_code_links":0,"syntology":null},{"paper":null,"slug":"reducing-estimation-bias-via-weighted-delayed","title":"WD3: Taming the Estimation Bias in Deep Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.12622","n_code_links":0,"syntology":null},{"paper":"/paper/forgetful-experience-replay-in-hierarchical","slug":"forgetful-experience-replay-in-hierarchical","title":"Forgetful Experience Replay in Hierarchical Reinforcement Learning from Demonstrations","date":"2020-06-17","arxiv_id":"2006.09939","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-uncertainty","title":"Reinforcement Learning with Uncertainty Estimation for Tactical Decision-Making in Intersections","date":"2020-06-17","arxiv_id":"2006.09786","n_code_links":0,"syntology":null},{"paper":null,"slug":"least-squares-regression-with-markovian-data","title":"Least Squares Regression with Markovian Data: Fundamental Limits and Algorithms","date":"2020-06-16","arxiv_id":"2006.08916","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-control-of-robotic","title":"Reinforcement Learning Control of Robotic Knee with Human in the Loop by Flexible Policy Iteration","date":"2020-06-16","arxiv_id":"2006.09008","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-online-evolving-framework-for-advancing","title":"An online evolving framework for advancing reinforcement-learning based automated vehicle control","date":"2020-06-15","arxiv_id":"2006.08092","n_code_links":0,"syntology":null},{"paper":"/paper/learning-heuristic-selection-with-dynamic","slug":"learning-heuristic-selection-with-dynamic","title":"Learning Heuristic Selection with Dynamic Algorithm Configuration","date":"2020-06-15","arxiv_id":"2006.08246","n_code_links":1,"syntology":null},{"paper":null,"slug":"human-and-multi-agent-collaboration-in-a","title":"Human and Multi-Agent collaboration in a human-MARL teaming framework","date":"2020-06-12","arxiv_id":"2006.07301","n_code_links":0,"syntology":null},{"paper":null,"slug":"balancing-a-cartpole-system-with","title":"Balancing a CartPole System with Reinforcement Learning -- A Tutorial","date":"2020-06-08","arxiv_id":"2006.04938","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-algorithm-configuration-foundation-of","slug":"dynamic-algorithm-configuration-foundation-of","title":"Dynamic Algorithm Configuration: Foundation of a New Meta-Algorithmic Framework","date":"2020-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/plangan-model-based-planning-with-sparse","slug":"plangan-model-based-planning-with-sparse","title":"PlanGAN: Model-based Planning With Sparse Rewards and Multiple Goals","date":"2020-06-01","arxiv_id":"2006.00900","n_code_links":1,"syntology":null},{"paper":"/paper/manipulating-the-distributions-of-experience","slug":"manipulating-the-distributions-of-experience","title":"Manipulating the Distributions of Experience used for Self-Play Learning in Expert Iteration","date":"2020-05-30","arxiv_id":"2006.00283","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimization-driven-deep-reinforcement","title":"Optimization-driven Deep Reinforcement Learning for Robust Beamforming in IRS-assisted Wireless Communications","date":"2020-05-25","arxiv_id":"2005.11885","n_code_links":0,"syntology":null},{"paper":null,"slug":"experience-augmentation-boosting-and","title":"Experience Augmentation: Boosting and Accelerating Off-Policy Multi-Agent Reinforcement Learning","date":"2020-05-19","arxiv_id":"2005.09453","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-turbulence-modeling-by-multi-agent","title":"Automating Turbulence Modeling by Multi-Agent Reinforcement Learning","date":"2020-05-18","arxiv_id":"2005.09023","n_code_links":0,"syntology":null},{"paper":null,"slug":"proxy-experience-replay-federated","title":"Proxy Experience Replay: Federated Distillation for Distributed Reinforcement Learning","date":"2020-05-13","arxiv_id":"2005.06105","n_code_links":0,"syntology":null},{"paper":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"toma-topological-map-abstraction-for","title":"TOMA: Topological Map Abstraction for Reinforcement Learning","date":"2020-05-11","arxiv_id":"2005.06061","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-fpga-based-on-device-reinforcement","title":"An FPGA-Based On-Device Reinforcement Learning Approach using Online Sequential Learning","date":"2020-05-10","arxiv_id":"2005.04646","n_code_links":0,"syntology":null},{"paper":"/paper/discrete-to-deep-supervised-policy-learning","slug":"discrete-to-deep-supervised-policy-learning","title":"Discrete-to-Deep Supervised Policy Learning","date":"2020-05-05","arxiv_id":"2005.02057","n_code_links":1,"syntology":null},{"paper":null,"slug":"delay-aware-resource-allocation-in-fog","title":"Delay-aware Resource Allocation in Fog-assisted IoT Networks Through Reinforcement Learning","date":"2020-04-30","arxiv_id":"2005.04097","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributional-soft-actor-critic-for-risk","title":"DSAC: Distributional Soft Actor Critic for Risk-Sensitive Reinforcement Learning","date":"2020-04-30","arxiv_id":"2004.14547","n_code_links":0,"syntology":null},{"paper":"/paper/reinforcement-learning-with-augmented-data","slug":"reinforcement-learning-with-augmented-data","title":"Reinforcement Learning with Augmented Data","date":"2020-04-30","arxiv_id":"2004.14990","n_code_links":2,"syntology":{"ran":21,"of":24,"n_ran_checked":9,"n_instrument":12,"unverified":3,"pointer_only":21,"phrase":"21 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 12 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"pbcs-efficient-exploration-and-exploitation","title":"PBCS : Efficient Exploration and Exploitation Using a Synergy between Reinforcement Learning and Motion Planning","date":"2020-04-24","arxiv_id":"2004.11667","n_code_links":0,"syntology":null},{"paper":"/paper/per-step-reward-a-new-perspective-for-risk","slug":"per-step-reward-a-new-perspective-for-risk","title":"Mean-Variance Policy Iteration for Risk-Averse Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10888","n_code_links":1,"syntology":null},{"paper":null,"slug":"stdpg-a-spatio-temporal-deterministic-policy","title":"STDPG: A Spatio-Temporal Deterministic Policy Gradient Agent for Dynamic Routing in SDN","date":"2020-04-21","arxiv_id":"2004.09783","n_code_links":0,"syntology":null},{"paper":"/paper/dark-experience-for-general-continual","slug":"dark-experience-for-general-continual","title":"Dark Experience for General Continual Learning: a Strong, Simple Baseline","date":"2020-04-15","arxiv_id":"2004.07211","n_code_links":3,"syntology":null},{"paper":null,"slug":"model-based-actor-critic-gan-drl-actor-critic","title":"Model-based actor-critic: GAN (model generator) + DRL (actor-critic) => AGI","date":"2020-04-04","arxiv_id":"2004.04574","n_code_links":0,"syntology":null},{"paper":"/paper/obstacle-avoidance-and-navigation-utilizing","slug":"obstacle-avoidance-and-navigation-utilizing","title":"Obstacle Avoidance and Navigation Utilizing Reinforcement Learning with Reward Shaping","date":"2020-03-28","arxiv_id":"2003.12863","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-agent-reinforcement-learning-for-1","title":"Multi-Agent Reinforcement Learning for Problems with Combined Individual and Team Reward","date":"2020-03-24","arxiv_id":"2003.10598","n_code_links":0,"syntology":null},{"paper":"/paper/evolutionary-population-curriculum-for-1","slug":"evolutionary-population-curriculum-for-1","title":"Evolutionary Population Curriculum for Scaling Multi-Agent Reinforcement Learning","date":"2020-03-23","arxiv_id":"2003.10423","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qian18long/epciclr2020"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"continual-graph-learning","title":"Overcoming Catastrophic Forgetting in Graph Neural Networks with Experience Replay","date":"2020-03-22","arxiv_id":"2003.09908","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-deep-reinforcement-learning-with","title":"Accelerating Deep Reinforcement Learning With the Aid of Partial Model: Energy-Efficient Predictive Video Streaming","date":"2020-03-21","arxiv_id":"2003.09708","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-continual-learning","slug":"adversarial-continual-learning","title":"Adversarial Continual Learning","date":"2020-03-21","arxiv_id":"2003.09553","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":4,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/Adversarial-Continual-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"online-continual-learning-on-sequences","title":"Online Continual Learning on Sequences","date":"2020-03-20","arxiv_id":"2003.09114","n_code_links":0,"syntology":null},{"paper":"/paper/robust-deep-reinforcement-learning-against","slug":"robust-deep-reinforcement-learning-against","title":"Robust Deep Reinforcement Learning against Adversarial Perturbations on State Observations","date":"2020-03-19","arxiv_id":"2003.08938","n_code_links":4,"syntology":null},{"paper":"/paper/particle-based-adaptive-discretization-for","slug":"particle-based-adaptive-discretization-for","title":"PFPN: Continuous Control of Physically Simulated Characters using Particle Filtering Policy Network","date":"2020-03-16","arxiv_id":"2003.06959","n_code_links":1,"syntology":null},{"paper":"/paper/deep-multi-agent-reinforcement-learning-for","slug":"deep-multi-agent-reinforcement-learning-for","title":"FACMAC: Factored Multi-Agent Centralised Policy Gradients","date":"2020-03-14","arxiv_id":"2003.06709","n_code_links":3,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["schroederdewitt/multiagent_mujoco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sample-efficient-reinforcement-learning","slug":"sample-efficient-reinforcement-learning","title":"Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","date":"2020-03-12","arxiv_id":"2003.06066","n_code_links":3,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/online-meta-critic-learning-for-off-policy-1","slug":"online-meta-critic-learning-for-off-policy-1","title":"Online Meta-Critic Learning for Off-Policy Actor-Critic Methods","date":"2020-03-11","arxiv_id":"2003.05334","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","official":{"repos":["zwfightzw/Meta-Critic"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamic-experience-replay","title":"Dynamic Experience Replay","date":"2020-03-04","arxiv_id":"2003.02372","n_code_links":0,"syntology":null},{"paper":"/paper/contention-window-optimization-in-ieee","slug":"contention-window-optimization-in-ieee","title":"Contention Window Optimization in IEEE 802.11ax Networks with Deep Reinforcement Learning","date":"2020-03-03","arxiv_id":"2003.01492","n_code_links":1,"syntology":null},{"paper":"/paper/reinforcement-co-learning-of-deep-and-spiking","slug":"reinforcement-co-learning-of-deep-and-spiking","title":"Reinforcement co-Learning of Deep and Spiking Neural Networks for Energy-Efficient Mapless Navigation with Neuromorphic Hardware","date":"2020-03-02","arxiv_id":"2003.01157","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["combra-lab/spiking-ddpg-mapless-navigation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-tuning-deep-reinforcement-learning","title":"A Self-Tuning Actor-Critic Algorithm","date":"2020-02-28","arxiv_id":"2002.12928","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-intelligent","title":"Deep Reinforcement Learning Based Intelligent Reflecting Surface for Secure Wireless Communications","date":"2020-02-27","arxiv_id":"2002.12271","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-freshness-and-energy-efficient-uav","title":"Data Freshness and Energy-Efficient UAV Navigation Optimization: A Deep Reinforcement Learning Approach","date":"2020-02-21","arxiv_id":"2003.04816","n_code_links":0,"syntology":null},{"paper":null,"slug":"disentangling-controllable-object-through","title":"Disentangling Controllable Object through Video Prediction Improves Visual Reinforcement Learning","date":"2020-02-21","arxiv_id":"2002.09136","n_code_links":0,"syntology":null},{"paper":"/paper/using-hindsight-to-anchor-past-knowledge-in-1","slug":"using-hindsight-to-anchor-past-knowledge-in-1","title":"Using Hindsight to Anchor Past Knowledge in Continual Learning","date":"2020-02-19","arxiv_id":"2002.08165","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-experience-selection-for-policy","title":"Adaptive Experience Selection for Policy Gradient","date":"2020-02-17","arxiv_id":"2002.06946","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-reinforcement-learning-for-anti-jamming","title":"Fast Reinforcement Learning for Anti-jamming Communications","date":"2020-02-13","arxiv_id":"2002.05364","n_code_links":0,"syntology":null},{"paper":null,"slug":"xcs-classifier-system-with-experience-replay","title":"XCS Classifier System with Experience Replay","date":"2020-02-13","arxiv_id":"2002.05628","n_code_links":0,"syntology":null},{"paper":"/paper/soft-hindsight-experience-replay","slug":"soft-hindsight-experience-replay","title":"Soft Hindsight Experience Replay","date":"2020-02-06","arxiv_id":"2002.02089","n_code_links":2,"syntology":null},{"paper":null,"slug":"bootstrapping-a-dqn-replay-memory-with","title":"Bootstrapping a DQN Replay Memory with Synthetic Experiences","date":"2020-02-04","arxiv_id":"2002.01370","n_code_links":0,"syntology":null},{"paper":null,"slug":"stacked-auto-encoder-based-deep-reinforcement","title":"Stacked Auto Encoder Based Deep Reinforcement Learning for Online Resource Scheduling in Large-Scale MEC Networks","date":"2020-01-24","arxiv_id":"2001.09223","n_code_links":0,"syntology":null},{"paper":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","n_code_links":4,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cooperative-highway-work-zone-merge-control","title":"Cooperative Highway Work Zone Merge Control based on Reinforcement Learning in A Connected and Automated Environment","date":"2020-01-21","arxiv_id":"2001.08581","n_code_links":0,"syntology":null},{"paper":"/paper/discriminator-soft-actor-critic-without","slug":"discriminator-soft-actor-critic-without","title":"Discriminator Soft Actor Critic without Extrinsic Rewards","date":"2020-01-19","arxiv_id":"2001.06808","n_code_links":1,"syntology":null},{"paper":null,"slug":"effects-of-sparse-rewards-of-different","title":"Effects of sparse rewards of different magnitudes in the speed of learning of model-based actor critic methods","date":"2020-01-18","arxiv_id":"2001.06725","n_code_links":0,"syntology":null},{"paper":"/paper/continuous-action-reinforcement-learning-for","slug":"continuous-action-reinforcement-learning-for","title":"Continuous-action Reinforcement Learning for Playing Racing Games: Comparing SPG to PPO","date":"2020-01-15","arxiv_id":"2001.05270","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-complex-1","title":"Deep Reinforcement Learning for Complex Manipulation Tasks with Sparse Feedback","date":"2020-01-12","arxiv_id":"2001.03877","n_code_links":0,"syntology":null},{"paper":"/paper/reward-engineering-for-object-pick-and-place","slug":"reward-engineering-for-object-pick-and-place","title":"Reward Engineering for Object Pick and Place Training","date":"2020-01-11","arxiv_id":"2001.03792","n_code_links":1,"syntology":null},{"paper":"/paper/population-guided-parallel-policy-search-for-1","slug":"population-guided-parallel-policy-search-for-1","title":"Population-Guided Parallel Policy Search for Reinforcement Learning","date":"2020-01-09","arxiv_id":"2001.02907","n_code_links":1,"syntology":null},{"paper":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/discrete-and-continuous-action-representation","slug":"discrete-and-continuous-action-representation","title":"Discrete and Continuous Action Representation for Practical RL in Video Games","date":"2019-12-23","arxiv_id":"1912.11077","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"exploiting-the-potential-of-deep","title":"Exploiting the potential of deep reinforcement learning for classification tasks in high-dimensional and unstructured data","date":"2019-12-20","arxiv_id":"1912.09595","n_code_links":0,"syntology":null},{"paper":null,"slug":"recruitment-imitation-mechanism-for","title":"Recruitment-imitation Mechanism for Evolutionary Reinforcement Learning","date":"2019-12-13","arxiv_id":"1912.06310","n_code_links":0,"syntology":null}],"record_sha256":"85c67d0d7c328a957caf0ecd53c57f87deb6b0bfab8437917c83bc1df9cdb024","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}