{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/5","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":9,"rows_per_page":100,"rows":[401,500],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay/papers/4","next":"/method/experience-replay/papers/6","papers":[{"paper":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","n_code_links":1,"syntology":null},{"paper":null,"slug":"decision-making-of-emergent-incident-based-on","title":"Decision-making of Emergent Incident based on P-MADDPG","date":"2022-03-19","arxiv_id":"2203.12673","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-infer-belief-embedded","title":"Learning to Infer Belief Embedded Communication","date":"2022-03-15","arxiv_id":"2203.07832","n_code_links":0,"syntology":null},{"paper":null,"slug":"lifelong-adaptive-machine-learning-for-sensor","title":"Lifelong Adaptive Machine Learning for Sensor-based Human Activity Recognition Using Prototypical Networks","date":"2022-03-11","arxiv_id":"2203.05692","n_code_links":0,"syntology":null},{"paper":null,"slug":"random-ensemble-reinforcement-learning-for","title":"Random Ensemble Reinforcement Learning for Traffic Signal Control","date":"2022-03-10","arxiv_id":"2203.05961","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-reinforcement-learning-for-predictive","title":"Designing Heterogeneous GNNs with Desired Permutation Properties for Wireless Resource Allocation","date":"2022-03-08","arxiv_id":"2203.03906","n_code_links":0,"syntology":null},{"paper":"/paper/new-insights-on-reducing-abrupt-1","slug":"new-insights-on-reducing-abrupt-1","title":"New Insights on Reducing Abrupt Representation Change in Online Continual Learning","date":"2022-03-08","arxiv_id":"2203.03798","n_code_links":2,"syntology":null},{"paper":"/paper/learning-bayesian-sparse-networks-with-full","slug":"learning-bayesian-sparse-networks-with-full","title":"Learning Bayesian Sparse Networks with Full Experience Replay for Continual Learning","date":"2022-02-21","arxiv_id":"2202.10203","n_code_links":0,"syntology":null},{"paper":"/paper/soft-actor-critic-deep-reinforcement-learning","slug":"soft-actor-critic-deep-reinforcement-learning","title":"Soft Actor-Critic Deep Reinforcement Learning for Fault Tolerant Flight Control","date":"2022-02-16","arxiv_id":"2202.09262","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-based-robust-resource-allocation-in-end-to","title":"AI-based Robust Resource Allocation in End-to-End Network Slicing under Demand and CSI Uncertainties","date":"2022-02-10","arxiv_id":"2202.05131","n_code_links":0,"syntology":null},{"paper":"/paper/imitation-learning-by-state-only-distribution","slug":"imitation-learning-by-state-only-distribution","title":"Imitation Learning by State-Only Distribution Matching","date":"2022-02-09","arxiv_id":"2202.04332","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["FeMa42/soil-tdm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/skrl-modular-and-flexible-library-for","slug":"skrl-modular-and-flexible-library-for","title":"skrl: Modular and Flexible Library for Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03825","n_code_links":1,"syntology":null},{"paper":"/paper/learning-fast-learning-slow-a-general-1","slug":"learning-fast-learning-slow-a-general-1","title":"Learning Fast, Learning Slow: A General Continual Learning Method based on Complementary Learning System","date":"2022-01-29","arxiv_id":"2201.12604","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["NeurAI-Lab/CLS-ER"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ddpg-driven-deep-unfolding-with-adaptive","title":"DDPG-Driven Deep-Unfolding with Adaptive Depth for Channel Estimation with Sparse Bayesian Learning","date":"2022-01-20","arxiv_id":"2201.08477","n_code_links":0,"syntology":null},{"paper":null,"slug":"critic-algorithms-using-cooperative-networks","title":"Critic Algorithms using Cooperative Networks","date":"2022-01-19","arxiv_id":"2201.07839","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-reinforcement-learning-algorithm","title":"An Improved Reinforcement Learning Algorithm for Learning to Branch","date":"2022-01-17","arxiv_id":"2201.06213","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-based-air-combat","title":"Reinforcement Learning based Air Combat Maneuver Generation","date":"2022-01-14","arxiv_id":"2201.05528","n_code_links":0,"syntology":null},{"paper":"/paper/smart-magnetic-microrobots-learn-to-swim-with","slug":"smart-magnetic-microrobots-learn-to-swim-with","title":"Smart Magnetic Microrobots Learn to Swim with Deep Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05599","n_code_links":1,"syntology":null},{"paper":null,"slug":"benchmarking-deep-reinforcement-learning","title":"Benchmarking Deep Reinforcement Learning Algorithms for Vision-based Robotics","date":"2022-01-11","arxiv_id":"2201.04224","n_code_links":0,"syntology":null},{"paper":null,"slug":"asymptotic-convergence-of-deep-multi-agent","title":"3DPG: Distributed Deep Deterministic Policy Gradient Algorithms for Networked Multi-Agent Systems","date":"2022-01-03","arxiv_id":"2201.00570","n_code_links":0,"syntology":null},{"paper":"/paper/class-incremental-continual-learning-into-the","slug":"class-incremental-continual-learning-into-the","title":"Class-Incremental Continual Learning into the eXtended DER-verse","date":"2022-01-03","arxiv_id":"2201.00766","n_code_links":1,"syntology":null},{"paper":null,"slug":"toward-pareto-efficient-fairness-utility","title":"Toward Pareto Efficient Fairness-Utility Trade-off inRecommendation through Reinforcement Learning","date":"2022-01-01","arxiv_id":"2201.00140","n_code_links":0,"syntology":null},{"paper":null,"slug":"actor-loss-of-soft-actor-critic-explained","title":"Actor Loss of Soft Actor Critic Explained","date":"2021-12-31","arxiv_id":"2112.15568","n_code_links":0,"syntology":null},{"paper":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-the-efficiency-of-off-policy","title":"Improving the Efficiency of Off-Policy Reinforcement Learning by Accounting for Past Decisions","date":"2021-12-23","arxiv_id":"2112.12281","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-optimal-power","title":"Deep Reinforcement Learning for Optimal Power Flow with Renewables Using Graph Information","date":"2021-12-22","arxiv_id":"2112.11461","n_code_links":0,"syntology":null},{"paper":null,"slug":"proving-theorems-using-incremental-learning-1","title":"Proving Theorems using Incremental Learning and Hindsight Experience Replay","date":"2021-12-20","arxiv_id":"2112.10664","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-reward-machines-a-study-in-partially","title":"Learning Reward Machines: A Study in Partially Observable Reinforcement Learning","date":"2021-12-17","arxiv_id":"2112.09477","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-results-for-q-learning-with","title":"Convergence Results For Q-Learning With Experience Replay","date":"2021-12-08","arxiv_id":"2112.04213","n_code_links":0,"syntology":null},{"paper":null,"slug":"hyper-parameter-optimization-based-on-soft","title":"Hyper-parameter optimization based on soft actor critic and hierarchical mixture regularization","date":"2021-12-08","arxiv_id":"2112.04084","n_code_links":0,"syntology":null},{"paper":null,"slug":"replay-for-safety","title":"Replay For Safety","date":"2021-12-08","arxiv_id":"2112.04229","n_code_links":0,"syntology":null},{"paper":null,"slug":"lifelong-domain-adaptation-via-consolidated","title":"Lifelong Domain Adaptation via Consolidated Internal Distribution","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-experience-replay-with-successor","title":"Improving Experience Replay with Successor Representation","date":"2021-11-29","arxiv_id":"2111.14331","n_code_links":0,"syntology":null},{"paper":"/paper/adaptively-calibrated-critic-estimates-for","slug":"adaptively-calibrated-critic-estimates-for","title":"Adaptively Calibrated Critic Estimates for Deep Reinforcement Learning","date":"2021-11-24","arxiv_id":"2111.12673","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nicolinho/acc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/generalized-decision-transformer-for-offline","slug":"generalized-decision-transformer-for-offline","title":"Generalized Decision Transformer for Offline Hindsight Information Matching","date":"2021-11-19","arxiv_id":"2111.10364","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["frt03/generalized_dt"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"improving-learning-from-demonstrations-by-1","title":"Improving Learning from Demonstrations by Learning from Experience","date":"2021-11-16","arxiv_id":"2111.08156","n_code_links":0,"syntology":null},{"paper":null,"slug":"power-norm-based-lifelong-learning-for","title":"Power Norm Based Lifelong Learning for Paraphrase Generations","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"awd3-dynamic-reduction-of-the-estimation-bias","title":"AWD3: Dynamic Reduction of the Estimation Bias","date":"2021-11-12","arxiv_id":"2111.06780","n_code_links":0,"syntology":null},{"paper":"/paper/improving-experience-replay-through-modeling","slug":"improving-experience-replay-through-modeling","title":"Improving Experience Replay through Modeling of Similar Transitions' Sets","date":"2021-11-12","arxiv_id":"2111.06907","n_code_links":1,"syntology":null},{"paper":null,"slug":"off-policy-correction-for-deep-deterministic","title":"Off-Policy Correction for Deep Deterministic Policy Gradient Algorithms via Batch Prioritized Experience Replay","date":"2021-11-02","arxiv_id":"2111.01865","n_code_links":0,"syntology":null},{"paper":null,"slug":"decentralized-multi-agent-reinforcement-3","title":"Decentralized Multi-Agent Reinforcement Learning: An Off-Policy Method","date":"2021-10-31","arxiv_id":"2111.00438","n_code_links":0,"syntology":null},{"paper":"/paper/hindsight-goal-ranking-on-replay-buffer-for","slug":"hindsight-goal-ranking-on-replay-buffer-for","title":"Hindsight Goal Ranking on Replay Buffer for Sparse Reward Environment","date":"2021-10-28","arxiv_id":"2110.15043","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-dpdk-based-acceleration-method-for","title":"Accelerating Distributed Deep Reinforcement Learning by In-Network Experience Sampling","date":"2021-10-26","arxiv_id":"2110.13506","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-control-of-overestimation-bias-for","title":"Automating Control of Overestimation Bias for Reinforcement Learning","date":"2021-10-26","arxiv_id":"2110.13523","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-off-policy-baselines-for-memory","slug":"recurrent-off-policy-baselines-for-memory","title":"Recurrent Off-policy Baselines for Memory-based Continuous Control","date":"2021-10-25","arxiv_id":"2110.12628","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-distributed-deep-reinforcement-learning","title":"A Distributed Deep Reinforcement Learning Technique for Application Placement in Edge and Fog Computing Environments","date":"2021-10-24","arxiv_id":"2110.12415","n_code_links":0,"syntology":null},{"paper":null,"slug":"locality-sensitive-experience-replay-for","title":"Locality-Sensitive Experience Replay for Online Recommendation","date":"2021-10-21","arxiv_id":"2110.10850","n_code_links":0,"syntology":null},{"paper":null,"slug":"computationally-efficient-safe-reinforcement","title":"Computationally Efficient Safe Reinforcement Learning for Power Systems","date":"2021-10-20","arxiv_id":"2110.10333","n_code_links":0,"syntology":null},{"paper":"/paper/green-simulation-assisted-policy-gradient-to","slug":"green-simulation-assisted-policy-gradient-to","title":"Variance Reduction based Experience Replay for Policy Optimization","date":"2021-10-17","arxiv_id":"2110.08902","n_code_links":1,"syntology":null},{"paper":null,"slug":"online-target-q-learning-with-reverse-1","title":"Online Target Q-learning with Reverse Experience Replay: Efficiently finding the Optimal Policy for Linear MDPs","date":"2021-10-16","arxiv_id":"2110.08440","n_code_links":0,"syntology":null},{"paper":null,"slug":"cognitively-inspired-learning-of-incremental-1","title":"Cognitively Inspired Learning of Incremental Drifting Concepts","date":"2021-10-09","arxiv_id":"2110.04662","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-aware-robot-navigation-via","title":"Human-Aware Robot Navigation via Reinforcement Learning with Hindsight Experience Replay and Curriculum Learning","date":"2021-10-09","arxiv_id":"2110.04564","n_code_links":0,"syntology":null},{"paper":null,"slug":"robotic-lever-manipulation-using-hindsight","title":"Robotic Lever Manipulation using Hindsight Experience Replay and Shapley Additive Explanations","date":"2021-10-07","arxiv_id":"2110.03292","n_code_links":0,"syntology":null},{"paper":null,"slug":"imaginary-hindsight-experience-replay-curious","title":"Imaginary Hindsight Experience Replay: Curious Model-based Learning for Sparse Reward Tasks","date":"2021-10-05","arxiv_id":"2110.02414","n_code_links":0,"syntology":null},{"paper":"/paper/hit-and-lead-discovery-with-explorative-rl","slug":"hit-and-lead-discovery-with-explorative-rl","title":"Hit and Lead Discovery with Explorative RL and Fragment-based Molecule Generation","date":"2021-10-04","arxiv_id":"2110.01219","n_code_links":0,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/large-batch-experience-replay","slug":"large-batch-experience-replay","title":"Large Batch Experience Replay","date":"2021-10-04","arxiv_id":"2110.01528","n_code_links":2,"syntology":{"ran":7,"of":11,"n_ran_checked":4,"n_instrument":3,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["sureli/laber"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"parallel-actors-and-learners-a-framework-for","title":"Parallel Actors and Learners: A Framework for Generating Scalable RL Implementations","date":"2021-10-03","arxiv_id":"2110.01101","n_code_links":0,"syntology":null},{"paper":"/paper/real-robot-challenge-using-deep-reinforcement","slug":"real-robot-challenge-using-deep-reinforcement","title":"Solving the Real Robot Challenge using Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.15233","n_code_links":2,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["robertmccarthy97/rrc_phase1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmarking-sample-selection-strategies-for","title":"Benchmarking Sample Selection Strategies for Batch Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"bootstrapped-hindsight-experience-replay-with","title":"Bootstrapped Hindsight Experience replay with Counterintuitive Prioritization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distributional-decision-transformer-for","title":"Distributional Decision Transformer for Hindsight Information Matching","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"experience-replay-more-when-it-s-a-key","title":"Experience Replay More When It's a Key Transition in Deep Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/explanation-aware-experience-replay-in-rule","slug":"explanation-aware-experience-replay-in-rule","title":"Explanation-Aware Experience Replay in Rule-Dense Environments","date":"2021-09-29","arxiv_id":"2109.14711","n_code_links":1,"syntology":null},{"paper":null,"slug":"faster-reinforcement-learning-with-value","title":"Faster Reinforcement Learning with Value Target Lower Bounding","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"goal-directed-planning-via-hindsight","title":"Goal-Directed Planning via Hindsight Experience Replay","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-imbalance-and-solution-in-online","title":"Gradient Imbalance and solution in Online Continual learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-meta-continual-learning","slug":"improving-meta-continual-learning","title":"Improving Meta-Continual Learning Representations with Representation Replay","date":"2021-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"meta-attention-for-off-policy-actor-critic","title":"Meta Attention For Off-Policy Actor-Critic","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"model-augmented-prioritized-experience-replay","title":"Model-augmented Prioritized Experience Replay","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-and-data-efficient-q-learning-by","title":"Robust and Data-efficient Q-learning by Composite Value-estimation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"splid-self-imitation-policy-learning-through","title":"SPLID: Self-Imitation Policy Learning through Iterative Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"spp-rl-state-planning-policy-reinforcement","title":"SPP-RL: State Planning Policy Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"superior-performance-with-diversified","title":"Superior Performance with Diversified Strategic Control in FPS Games Using General Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/prioritized-experience-based-reinforcement","slug":"prioritized-experience-based-reinforcement","title":"Prioritized Experience-based Reinforcement Learning with Human Guidance for Autonomous Driving","date":"2021-09-26","arxiv_id":"2109.12516","n_code_links":1,"syntology":null},{"paper":null,"slug":"improved-soft-actor-critic-mixing-prioritized","title":"Improved Soft Actor-Critic: Mixing Prioritized Off-Policy Samples with On-Policy Experience","date":"2021-09-24","arxiv_id":"2109.11767","n_code_links":0,"syntology":null},{"paper":null,"slug":"pranet-point-cloud-registration-with-an","title":"PRANet: Point Cloud Registration with an Artificial Agent","date":"2021-09-23","arxiv_id":"2109.11349","n_code_links":0,"syntology":null},{"paper":"/paper/trust-region-policy-optimisation-in-multi","slug":"trust-region-policy-optimisation-in-multi","title":"Trust Region Policy Optimisation in Multi-Agent Reinforcement Learning","date":"2021-09-23","arxiv_id":"2109.11251","n_code_links":11,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cyanrain7/trust-region-policy-optimisation-in-multi-agent-reinforcement-learning","anonymous-iclr22/trust-region-in-multi-agent-reinforcement-learning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"towards-multi-agent-reinforcement-learning-1","title":"Towards Multi-Agent Reinforcement Learning using Quantum Boltzmann Machines","date":"2021-09-22","arxiv_id":"2109.10900","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-based-2","title":"Deep Reinforcement Learning Based Multidimensional Resource Management for Energy Harvesting Cognitive NOMA Communications","date":"2021-09-17","arxiv_id":"2109.09503","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-learning-with-sparse-experience-replay-1","title":"Meta-Learning with Sparse Experience Replay for Lifelong Language Learning","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dromo-distributionally-robust-offline-model","title":"DROMO: Distributionally Robust Offline Model-based Policy Optimization","date":"2021-09-15","arxiv_id":"2109.07275","n_code_links":0,"syntology":null},{"paper":null,"slug":"ader-adapting-between-exploration-and","title":"ADER:Adapting between Exploration and Robustness for Actor-Critic Methods","date":"2021-09-08","arxiv_id":"2109.03443","n_code_links":0,"syntology":null},{"paper":"/paper/macrpo-multi-agent-cooperative-recurrent","slug":"macrpo-multi-agent-cooperative-recurrent","title":"MACRPO: Multi-Agent Cooperative Recurrent Policy Optimization","date":"2021-09-02","arxiv_id":"2109.00882","n_code_links":1,"syntology":null},{"paper":null,"slug":"path-planning-for-cellular-connected-uav-a","title":"Path Planning for Cellular-Connected UAV: A DRL Solution with Quantum-Inspired Experience Replay","date":"2021-08-30","arxiv_id":"2108.13184","n_code_links":0,"syntology":null},{"paper":null,"slug":"detection-and-continual-learning-of-novel","title":"Detection and Continual Learning of Novel Face Presentation Attacks","date":"2021-08-27","arxiv_id":"2108.12081","n_code_links":0,"syntology":null},{"paper":null,"slug":"wad-a-deep-reinforcement-learning-agent-for","title":"WAD: A Deep Reinforcement Learning Agent for Urban Autonomous Driving","date":"2021-08-27","arxiv_id":"2108.12134","n_code_links":0,"syntology":null},{"paper":"/paper/responsive-regulation-of-dynamic-uav","slug":"responsive-regulation-of-dynamic-uav","title":"Responsive Regulation of Dynamic UAV Communication Networks Based on Deep Reinforcement Learning","date":"2021-08-25","arxiv_id":"2108.11012","n_code_links":1,"syntology":null},{"paper":null,"slug":"principal-gradient-direction-and-confidence","title":"Principal Gradient Direction and Confidence Reservoir Sampling for Continual Learning","date":"2021-08-21","arxiv_id":"2108.09592","n_code_links":0,"syntology":null},{"paper":"/paper/diversity-based-trajectory-and-goal-selection","slug":"diversity-based-trajectory-and-goal-selection","title":"Diversity-based Trajectory and Goal Selection with Hindsight Experience Replay","date":"2021-08-17","arxiv_id":"2108.07887","n_code_links":1,"syntology":null},{"paper":null,"slug":"optimal-scheduling-of-isolated-microgrids","title":"Optimal Scheduling of Isolated Microgrids Using Automated Reinforcement Learning-based Multi-period Forecasting","date":"2021-08-15","arxiv_id":"2108.06764","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-neural-mapping-learning-an-implicit","title":"Continual Neural Mapping: Learning An Implicit Scene Representation from Sequential Observations","date":"2021-08-12","arxiv_id":"2108.05851","n_code_links":0,"syntology":null},{"paper":null,"slug":"modified-double-dqn-addressing-stability","title":"Modified Double DQN: addressing stability","date":"2021-08-09","arxiv_id":"2108.04115","n_code_links":0,"syntology":null},{"paper":"/paper/safe-deep-reinforcement-learning-for-multi","slug":"safe-deep-reinforcement-learning-for-multi","title":"Safe Deep Reinforcement Learning for Multi-Agent Systems with Continuous Action Spaces","date":"2021-08-09","arxiv_id":"2108.03952","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zisikons/deep-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-reinforcement-learning-for-continuous","title":"Deep Reinforcement Learning for Continuous Docking Control of Autonomous Underwater Vehicles: A Benchmarking Study","date":"2021-08-05","arxiv_id":"2108.02665","n_code_links":0,"syntology":null},{"paper":"/paper/risk-conditioned-neural-motion-planning","slug":"risk-conditioned-neural-motion-planning","title":"Risk Conditioned Neural Motion Planning","date":"2021-08-04","arxiv_id":"2108.01851","n_code_links":1,"syntology":null},{"paper":null,"slug":"value-based-reinforcement-learning-for","title":"Value-Based Reinforcement Learning for Continuous Control Robotic Manipulation in Multi-Task Sparse Reward Settings","date":"2021-07-28","arxiv_id":"2107.13356","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-algorithm-of-robot-path-planning","title":"An Improved Algorithm of Robot Path Planning in Complex Environment Based on Double DQN","date":"2021-07-23","arxiv_id":"2107.11245","n_code_links":0,"syntology":null},{"paper":"/paper/a-reinforcement-learning-environment-for-2","slug":"a-reinforcement-learning-environment-for-2","title":"A Reinforcement Learning Environment for Mathematical Reasoning via Program Synthesis","date":"2021-07-15","arxiv_id":"2107.07373","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixing-human-demonstrations-with-self","title":"Mixing Human Demonstrations with Self-Exploration in Experience Replay for Deep Reinforcement Learning","date":"2021-07-14","arxiv_id":"2107.06840","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-for-6","title":"A Deep Reinforcement Learning Approach for Traffic Signal Control Optimization","date":"2021-07-13","arxiv_id":"2107.06115","n_code_links":0,"syntology":null}],"record_sha256":"f0a2c7131d8df093200f943e2d75ba9276340180eeeaa75427c3fa35d21cfc91","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}