{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/4","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":9,"rows_per_page":100,"rows":[301,400],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay/papers/3","next":"/method/experience-replay/papers/5","papers":[{"paper":null,"slug":"a-memory-efficient-deep-reinforcement","title":"A Memory Efficient Deep Reinforcement Learning Approach For Snake Game Autonomous Agents","date":"2023-01-27","arxiv_id":"2301.11977","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-deep-reinforcement-learning-based-2","title":"A Novel Deep Reinforcement Learning-based Approach for Enhancing Spectral Efficiency of IRS-assisted Wireless Systems","date":"2023-01-24","arxiv_id":"2302.14706","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-multi-agent-deep-deterministic-policy","title":"On Multi-Agent Deep Deterministic Policy Gradients and their Explainability for SMARTS Environment","date":"2023-01-20","arxiv_id":"2301.09420","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-deep-reinforcement-learning-for","title":"Automated deep reinforcement learning for real-time scheduling strategy of multi-energy system integrated with post-carbon and direct-air carbon captured system","date":"2023-01-18","arxiv_id":"2301.07768","n_code_links":0,"syntology":null},{"paper":null,"slug":"pdvn-a-patch-based-dual-view-network-for-face","title":"PDVN: A Patch-based Dual-view Network for Face Liveness Detection using Light Field Focal Stack","date":"2023-01-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/online-class-incremental-learning-for-real","slug":"online-class-incremental-learning-for-real","title":"Online Class-Incremental Learning For Real-World Food Image Classification","date":"2023-01-12","arxiv_id":"2301.05246","n_code_links":1,"syntology":null},{"paper":null,"slug":"actor-director-critic-a-novel-deep","title":"Actor-Director-Critic: A Novel Deep Reinforcement Learning Framework","date":"2023-01-10","arxiv_id":"2301.03887","n_code_links":0,"syntology":null},{"paper":"/paper/hint-assisted-reinforcement-learning-an","slug":"hint-assisted-reinforcement-learning-an","title":"Hint assisted reinforcement learning: an application in radio astronomy","date":"2023-01-10","arxiv_id":"2301.03933","n_code_links":1,"syntology":null},{"paper":"/paper/extreme-q-learning-maxent-rl-without-entropy","slug":"extreme-q-learning-maxent-rl-without-entropy","title":"Extreme Q-Learning: MaxEnt RL without Entropy","date":"2023-01-05","arxiv_id":"2301.02328","n_code_links":4,"syntology":{"ran":8,"of":13,"n_ran_checked":5,"n_instrument":3,"unverified":5,"pointer_only":7,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"on-the-geometry-of-reinforcement-learning-in","title":"On the Geometry of Reinforcement Learning in Continuous State and Action Spaces","date":"2022-12-29","arxiv_id":"2301.00009","n_code_links":0,"syntology":null},{"paper":"/paper/lifelong-reinforcement-learning-with","slug":"lifelong-reinforcement-learning-with","title":"Lifelong Reinforcement Learning with Modulating Masks","date":"2022-12-21","arxiv_id":"2212.11110","n_code_links":4,"syntology":null},{"paper":null,"slug":"neighboring-state-based-rl-exploration","title":"Neighboring state-based RL Exploration","date":"2022-12-21","arxiv_id":"2212.10712","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-exploration-in-resource-restricted","title":"Efficient Exploration in Resource-Restricted Reinforcement Learning","date":"2022-12-14","arxiv_id":"2212.06988","n_code_links":0,"syntology":null},{"paper":null,"slug":"off-policy-deep-reinforcement-learning-1","title":"Off-Policy Deep Reinforcement Learning Algorithms for Handling Various Robotic Manipulator Tasks","date":"2022-12-11","arxiv_id":"2212.05572","n_code_links":0,"syntology":null},{"paper":"/paper/td3-with-reverse-kl-regularizer-for-offline","slug":"td3-with-reverse-kl-regularizer-for-offline","title":"TD3 with Reverse KL Regularizer for Offline Reinforcement Learning from Mixed Datasets","date":"2022-12-05","arxiv_id":"2212.02125","n_code_links":1,"syntology":null},{"paper":null,"slug":"automatic-discovery-of-multi-perspective","title":"Automatic Discovery of Multi-perspective Process Model using Reinforcement Learning","date":"2022-11-30","arxiv_id":"2211.16687","n_code_links":0,"syntology":null},{"paper":"/paper/the-surprising-effectiveness-of-latent-world","slug":"the-surprising-effectiveness-of-latent-world","title":"The Effectiveness of World Models for Continual Reinforcement Learning","date":"2022-11-29","arxiv_id":"2211.15944","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-from-good-trajectories-in-offline","title":"Learning from Good Trajectories in Offline Multi-Agent Reinforcement Learning","date":"2022-11-28","arxiv_id":"2211.15612","n_code_links":0,"syntology":null},{"paper":null,"slug":"rl-based-guidance-in-outpatient-hysteroscopy","title":"RL-Based Guidance in Outpatient Hysteroscopy Training: A Feasibility Study","date":"2022-11-26","arxiv_id":"2211.14541","n_code_links":0,"syntology":null},{"paper":null,"slug":"double-deep-q-learning-in-opponent-modeling","title":"Double Deep Q-Learning in Opponent Modeling","date":"2022-11-24","arxiv_id":"2211.15384","n_code_links":0,"syntology":null},{"paper":null,"slug":"cacto-continuous-actor-critic-with-trajectory","title":"CACTO: Continuous Actor-Critic with Trajectory Optimization -- Towards global optimality","date":"2022-11-12","arxiv_id":"2211.06625","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-microgrid","title":"Deep Reinforcement Learning Microgrid Optimization Strategy Considering Priority Flexible Demand Side","date":"2022-11-11","arxiv_id":"2211.05946","n_code_links":0,"syntology":null},{"paper":"/paper/the-benefits-of-model-based-generalization-in","slug":"the-benefits-of-model-based-generalization-in","title":"The Benefits of Model-Based Generalization in Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02222","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kenjyoung/model_generalization_code_supplement"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"wind-power-forecasting-considering-data","title":"Wind Power Forecasting Considering Data Privacy Protection: A Federated Deep Reinforcement Learning Approach","date":"2022-11-02","arxiv_id":"2211.02674","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-contrastive-samples-for-identifying-and","title":"Using Contrastive Samples for Identifying and Leveraging Possible Causal Relationships in Reinforcement Learning","date":"2022-10-28","arxiv_id":"2210.17296","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-demonstrations-with-latent-space","slug":"leveraging-demonstrations-with-latent-space","title":"Leveraging Demonstrations with Latent Space Priors","date":"2022-10-26","arxiv_id":"2210.14685","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/latent-space-priors"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/aacher-assorted-actor-critic-deep","slug":"aacher-assorted-actor-critic-deep","title":"AACHER: Assorted Actor-Critic Deep Reinforcement Learning with Hindsight Experience Replay","date":"2022-10-24","arxiv_id":"2210.12892","n_code_links":1,"syntology":null},{"paper":"/paper/meet-a-monte-carlo-exploration-exploitation","slug":"meet-a-monte-carlo-exploration-exploitation","title":"MEET: A Monte Carlo Exploration-Exploitation Trade-off for Buffer Sampling","date":"2022-10-24","arxiv_id":"2210.13545","n_code_links":1,"syntology":null},{"paper":"/paper/navigating-memory-construction-by-global","slug":"navigating-memory-construction-by-global","title":"Navigating Memory Construction by Global Pseudo-Task Simulation for Continual Learning","date":"2022-10-16","arxiv_id":"2210.08442","n_code_links":1,"syntology":null},{"paper":"/paper/multiagent-reinforcement-learning-based-on","slug":"multiagent-reinforcement-learning-based-on","title":"Multiagent Reinforcement Learning Based on Fusion-Multiactor-Attention-Critic for Multiple-Unmanned-Aerial-Vehicle Navigation Control","date":"2022-10-10","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"algorithmic-trading-using-continuous-action","title":"Algorithmic Trading Using Continuous Action Space Deep Reinforcement Learning","date":"2022-10-07","arxiv_id":"2210.03469","n_code_links":0,"syntology":null},{"paper":null,"slug":"elastic-step-dqn-a-novel-multi-step-algorithm","title":"Elastic Step DQN: A novel multi-step algorithm to alleviate overestimation in Deep QNetworks","date":"2022-10-07","arxiv_id":"2210.03325","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-entropy-maximizing-td3-based","title":"A Novel Entropy-Maximizing TD3-based Reinforcement Learning for Automatic PID Tuning","date":"2022-10-05","arxiv_id":"2210.02381","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-relevant-is-selective-memory-population","title":"How Relevant is Selective Memory Population in Lifelong Language Learning?","date":"2022-10-03","arxiv_id":"2210.00940","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-hindsight-goal-relabeling","title":"Understanding Hindsight Goal Relabeling from a Divergence Minimization Perspective","date":"2022-09-26","arxiv_id":"2209.13046","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimizing-human-assistance-augmenting-a","title":"Minimizing Human Assistance: Augmenting a Single Demonstration for Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.11275","n_code_links":0,"syntology":null},{"paper":"/paper/continual-vqa-for-disaster-response-systems","slug":"continual-vqa-for-disaster-response-systems","title":"Continual VQA for Disaster Response Systems","date":"2022-09-21","arxiv_id":"2209.10320","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-based-charging","title":"A Deep Reinforcement Learning-Based Charging Scheduling Approach with Augmented Lagrangian for Electric Vehicle","date":"2022-09-20","arxiv_id":"2209.09772","n_code_links":0,"syntology":null},{"paper":"/paper/m-2-dqn-a-robust-method-for-accelerating-deep","slug":"m-2-dqn-a-robust-method-for-accelerating-deep","title":"M$^2$DQN: A Robust Method for Accelerating Deep Q-learning Network","date":"2022-09-16","arxiv_id":"2209.07809","n_code_links":1,"syntology":null},{"paper":null,"slug":"data-efficient-reinforcement-learning-and","title":"Data efficient reinforcement learning and adaptive optimal perimeter control of network traffic dynamics","date":"2022-09-13","arxiv_id":"2209.05726","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-continual-learning-via-the-meta","title":"Online Continual Learning via the Meta-learning Update with Multi-scale Knowledge Distillation and Data Augmentation","date":"2022-09-12","arxiv_id":"2209.06107","n_code_links":0,"syntology":null},{"paper":"/paper/partial-observability-during-drl-for-robot","slug":"partial-observability-during-drl-for-robot","title":"Experimental Study on The Effect of Multi-step Deep Reinforcement Learning in POMDPs","date":"2022-09-12","arxiv_id":"2209.04999","n_code_links":1,"syntology":null},{"paper":null,"slug":"selecting-related-knowledge-via-efficient","title":"Selecting Related Knowledge via Efficient Channel Attention for Online Continual Learning","date":"2022-09-09","arxiv_id":"2209.04212","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-approach-to-training-multiple","title":"A New Approach to Training Multiple Cooperative Agents for Autonomous Driving","date":"2022-09-05","arxiv_id":"2209.02157","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-continual-learning-for-graphs","title":"Reinforced Continual Learning for Graphs","date":"2022-09-04","arxiv_id":"2209.01556","n_code_links":0,"syntology":null},{"paper":"/paper/actor-prioritized-experience-replay","slug":"actor-prioritized-experience-replay","title":"Actor Prioritized Experience Replay","date":"2022-09-01","arxiv_id":"2209.00532","n_code_links":1,"syntology":null},{"paper":null,"slug":"cluster-based-sampling-in-hindsight","title":"Cluster-based Sampling in Hindsight Experience Replay for Robotic Tasks (Student Abstract)","date":"2022-08-31","arxiv_id":"2208.14741","n_code_links":0,"syntology":null},{"paper":null,"slug":"digital-twin-assisted-risk-aware-sleep-mode","title":"Digital Twin Assisted Risk-Aware Sleep Mode Management Using Deep Q-Networks","date":"2022-08-30","arxiv_id":"2208.14380","n_code_links":0,"syntology":null},{"paper":"/paper/goal-conditioned-q-learning-as-knowledge","slug":"goal-conditioned-q-learning-as-knowledge","title":"Goal-Conditioned Q-Learning as Knowledge Distillation","date":"2022-08-28","arxiv_id":"2208.13298","n_code_links":1,"syntology":null},{"paper":"/paper/variance-reduction-based-experience-replay","slug":"variance-reduction-based-experience-replay","title":"Variance Reduction based Experience Replay for Policy Optimization","date":"2022-08-25","arxiv_id":"2208.12341","n_code_links":1,"syntology":null},{"paper":"/paper/bsac-bayesian-strategy-network-based-soft","slug":"bsac-bayesian-strategy-network-based-soft","title":"Bayesian Soft Actor-Critic: A Directed Acyclic Strategy Graph Based Deep Reinforcement Learning","date":"2022-08-11","arxiv_id":"2208.06033","n_code_links":2,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-dynamic-4","title":"Plug-and-Play Model-Agnostic Counterfactual Policy Synthesis for Deep Reinforcement Learning based Recommendation","date":"2022-08-10","arxiv_id":"2208.05142","n_code_links":0,"syntology":null},{"paper":"/paper/performance-comparison-of-deep-rl-algorithms","slug":"performance-comparison-of-deep-rl-algorithms","title":"Performance Comparison of Deep RL Algorithms for Energy Systems Optimal Scheduling","date":"2022-08-01","arxiv_id":"2208.00728","n_code_links":1,"syntology":null},{"paper":"/paper/relay-hindsight-experience-replay-continual","slug":"relay-hindsight-experience-replay-continual","title":"Relay Hindsight Experience Replay: Self-Guided Continual Reinforcement Learning for Sequential Object Manipulation Tasks with Sparse Rewards","date":"2022-08-01","arxiv_id":"2208.00843","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributional-actor-critic-ensemble-for","title":"Distributional Actor-Critic Ensemble for Uncertainty-Aware Continuous Control","date":"2022-07-27","arxiv_id":"2207.13730","n_code_links":0,"syntology":null},{"paper":"/paper/safe-and-robust-experience-sharing-for","slug":"safe-and-robust-experience-sharing-for","title":"Safe and Robust Experience Sharing for Deterministic Policy Gradient Algorithms","date":"2022-07-27","arxiv_id":"2207.13453","n_code_links":1,"syntology":null},{"paper":"/paper/abstract-demonstrations-and-adaptive","slug":"abstract-demonstrations-and-adaptive","title":"Abstract Demonstrations and Adaptive Exploration for Efficient and Stable Multi-step Sparse Reward Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09243","n_code_links":1,"syntology":null},{"paper":null,"slug":"associative-memory-based-experience-replay","title":"Associative Memory Based Experience Replay for Deep Reinforcement Learning","date":"2022-07-16","arxiv_id":"2207.07791","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-agent-deep-reinforcement-learning-2","title":"Multi-Agent Deep Reinforcement Learning-Driven Mitigation of Adverse Effects of Cyber-Attacks on Electric Vehicle Charging Station","date":"2022-07-14","arxiv_id":"2207.07041","n_code_links":0,"syntology":null},{"paper":"/paper/consistency-is-the-key-to-further-mitigating","slug":"consistency-is-the-key-to-further-mitigating","title":"Consistency is the key to further mitigating catastrophic forgetting in continual learning","date":"2022-07-11","arxiv_id":"2207.04998","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["neurai-lab/consistencycl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"asynchronous-curriculum-experience-replay-a","title":"Asynchronous Curriculum Experience Replay: A Deep Reinforcement Learning Approach for UAV Autonomous Motion Control in Unknown Dynamic Environments","date":"2022-07-04","arxiv_id":"2207.01251","n_code_links":0,"syntology":null},{"paper":"/paper/it-s-all-about-consistency-a-study-on-memory","slug":"it-s-all-about-consistency-a-study-on-memory","title":"Memory Population in Continual Learning via Outlier Elimination","date":"2022-07-04","arxiv_id":"2207.01145","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["JuliousHurtado/MOE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"distributed-online-system-identification-for","title":"Distributed Online System Identification for LTI Systems Using Reverse Experience Replay","date":"2022-07-03","arxiv_id":"2207.01062","n_code_links":0,"syntology":null},{"paper":"/paper/usher-unbiased-sampling-for-hindsight","slug":"usher-unbiased-sampling-for-hindsight","title":"USHER: Unbiased Sampling for Hindsight Experience Replay","date":"2022-07-03","arxiv_id":"2207.01115","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","n_code_links":1,"syntology":null},{"paper":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"paper":"/paper/decentralized-distributed-learning-with","slug":"decentralized-distributed-learning-with","title":"FedER: Federated Learning through Experience Replay and Privacy-Preserving Data Synthesis","date":"2022-06-20","arxiv_id":"2206.10048","n_code_links":1,"syntology":null},{"paper":"/paper/maser-multi-agent-reinforcement-learning-with","slug":"maser-multi-agent-reinforcement-learning-with","title":"MASER: Multi-Agent Reinforcement Learning with Subgoals Generated from Experience Replay Buffer","date":"2022-06-20","arxiv_id":"2206.10607","n_code_links":1,"syntology":null},{"paper":null,"slug":"two-hop-age-of-information-scheduling-for","title":"Two-Hop Age of Information Scheduling for Multi-UAV Assisted Mobile Edge Computing: FRL vs MADDPG","date":"2022-06-19","arxiv_id":"2206.09488","n_code_links":0,"syntology":null},{"paper":null,"slug":"autonomous-platoon-control-with-integrated","title":"Autonomous Platoon Control with Integrated Deep Reinforcement Learning and Dynamic Programming","date":"2022-06-15","arxiv_id":"2206.07536","n_code_links":0,"syntology":null},{"paper":null,"slug":"computation-offloading-and-resource-1","title":"Computation Offloading and Resource Allocation in F-RANs: A Federated Deep Reinforcement Learning Approach","date":"2022-06-13","arxiv_id":"2206.05881","n_code_links":0,"syntology":null},{"paper":"/paper/synergy-between-synaptic-consolidation-and","slug":"synergy-between-synaptic-consolidation-and","title":"SYNERgy between SYNaptic consolidation and Experience Replay for general continual learning","date":"2022-06-08","arxiv_id":"2206.04016","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neurai-lab/synergy"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/look-back-when-surprised-stabilizing-reverse","slug":"look-back-when-surprised-stabilizing-reverse","title":"Introspective Experience Replay: Look Back When Surprised","date":"2022-06-07","arxiv_id":"2206.03171","n_code_links":1,"syntology":null},{"paper":null,"slug":"balancing-profit-risk-and-sustainability-for","title":"Balancing Profit, Risk, and Sustainability for Portfolio Management","date":"2022-06-06","arxiv_id":"2207.02134","n_code_links":0,"syntology":null},{"paper":null,"slug":"goal-space-planning-with-subgoal-models","title":"Goal-Space Planning with Subgoal Models","date":"2022-06-06","arxiv_id":"2206.02902","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-adversarial-attacks-detection-based-on","title":"Robust Adversarial Attacks Detection based on Explainable Deep Reinforcement Learning For UAV Guidance and Planning","date":"2022-06-06","arxiv_id":"2206.02670","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-energy-dispatch-and-unit-commitment-in","title":"Joint Energy Dispatch and Unit Commitment in Microgrids Based on Deep Reinforcement Learning","date":"2022-06-03","arxiv_id":"2206.01663","n_code_links":0,"syntology":null},{"paper":null,"slug":"equivariant-reinforcement-learning-for","title":"Equivariant Reinforcement Learning for Quadrotor UAV","date":"2022-06-02","arxiv_id":"2206.01233","n_code_links":0,"syntology":null},{"paper":null,"slug":"stabilizing-q-learning-with-linear","title":"Stabilizing Q-learning with Linear Architectures for Provably Efficient Learning","date":"2022-06-01","arxiv_id":"2206.00796","n_code_links":0,"syntology":null},{"paper":"/paper/truly-deterministic-policy-optimization-1","slug":"truly-deterministic-policy-optimization-1","title":"Truly Deterministic Policy Optimization","date":"2022-05-30","arxiv_id":"2205.15379","n_code_links":1,"syntology":null},{"paper":"/paper/task-agnostic-continual-reinforcement","slug":"task-agnostic-continual-reinforcement","title":"Task-Agnostic Continual Reinforcement Learning: Gaining Insights and Overcoming Challenges","date":"2022-05-28","arxiv_id":"2205.14495","n_code_links":2,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["amazon-science/replay-based-recurrent-rl","amazon-research/replay-based-recurrent-rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"multiple-domain-cyberspace-attack-and-defense","title":"Multiple Domain Cyberspace Attack and Defense Game Based on Reward Randomization Reinforcement Learning","date":"2022-05-23","arxiv_id":"2205.10990","n_code_links":0,"syntology":null},{"paper":"/paper/memory-efficient-reinforcement-learning-with","slug":"memory-efficient-reinforcement-learning-with","title":"Memory-efficient Reinforcement Learning with Value-based Knowledge Consolidation","date":"2022-05-22","arxiv_id":"2205.10868","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["qlan3/MeDQN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/dexterous-robotic-manipulation-using-deep","slug":"dexterous-robotic-manipulation-using-deep","title":"Dexterous Robotic Manipulation using Deep Reinforcement Learning and Knowledge Transfer for Complex Sparse Reward-based Tasks","date":"2022-05-19","arxiv_id":"2205.09683","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wq13552463699/rrc_pos_ori"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/neighborhood-mixup-experience-replay-local","slug":"neighborhood-mixup-experience-replay-local","title":"Neighborhood Mixup Experience Replay: Local Convex Interpolation for Improved Sample Efficiency in Continuous Control Tasks","date":"2022-05-18","arxiv_id":"2205.09117","n_code_links":1,"syntology":null},{"paper":"/paper/optimal-adaptive-prediction-intervals-for","slug":"optimal-adaptive-prediction-intervals-for","title":"Optimal Adaptive Prediction Intervals for Electricity Load Forecasting in Distribution Systems via Reinforcement Learning","date":"2022-05-18","arxiv_id":"2205.08698","n_code_links":1,"syntology":null},{"paper":null,"slug":"qhd-a-brain-inspired-hyperdimensional","title":"Efficient Off-Policy Reinforcement Learning via Brain-Inspired Computing","date":"2022-05-14","arxiv_id":"2205.06978","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-reinforcement-learning-for-star-riss-a","title":"Hybrid Reinforcement Learning for STAR-RISs: A Coupled Phase-Shift Model Based Beamformer","date":"2022-05-10","arxiv_id":"2205.05029","n_code_links":0,"syntology":null},{"paper":"/paper/variance-reduction-based-partial-trajectory","slug":"variance-reduction-based-partial-trajectory","title":"Variance Reduction based Partial Trajectory Reuse to Accelerate Policy Gradient Optimization","date":"2022-05-06","arxiv_id":"2205.02976","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-gaussian-mixture-critic-in-off","slug":"revisiting-gaussian-mixture-critic-in-off","title":"Revisiting Gaussian mixture critics in off-policy reinforcement learning: a sample-based approach","date":"2022-04-21","arxiv_id":"2204.10256","n_code_links":1,"syntology":null},{"paper":"/paper/continual-predictive-learning-from-videos","slug":"continual-predictive-learning-from-videos","title":"Continual Predictive Learning from Videos","date":"2022-04-12","arxiv_id":"2204.05624","n_code_links":1,"syntology":null},{"paper":"/paper/confidence-estimation-transformer-for-long","slug":"confidence-estimation-transformer-for-long","title":"Confidence Estimation Transformer for Long-term Renewable Energy Forecasting in Reinforcement Learning-based Power Grid Dispatching","date":"2022-04-10","arxiv_id":"2204.04612","n_code_links":1,"syntology":null},{"paper":null,"slug":"ma-dreamer-coordination-and-communication","title":"MA-Dreamer: Coordination and communication through shared imagination","date":"2022-04-10","arxiv_id":"2204.04687","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-exploration-from-language","title":"Semantic Exploration from Language Abstractions and Pretrained Representations","date":"2022-04-08","arxiv_id":"2204.05080","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-word-error-rate-a-good-evaluation-metric","title":"Is Word Error Rate a good evaluation metric for Speech Recognition in Indic Languages?","date":"2022-03-30","arxiv_id":"2203.16601","n_code_links":0,"syntology":null},{"paper":"/paper/topological-experience-replay-1","slug":"topological-experience-replay-1","title":"Topological Experience Replay","date":"2022-03-29","arxiv_id":"2203.15845","n_code_links":1,"syntology":null},{"paper":null,"slug":"collaborative-intelligent-reflecting-surface","title":"Collaborative Intelligent Reflecting Surface Networks with Multi-Agent Reinforcement Learning","date":"2022-03-26","arxiv_id":"2203.14152","n_code_links":0,"syntology":null},{"paper":null,"slug":"non-parametric-stochastic-policy-gradient","title":"Non-Parametric Stochastic Policy Gradient with Strategic Retreat for Non-Stationary Environment","date":"2022-03-24","arxiv_id":"2203.14905","n_code_links":0,"syntology":null},{"paper":null,"slug":"remember-and-forget-experience-replay-for","title":"Remember and Forget Experience Replay for Multi-Agent Reinforcement Learning","date":"2022-03-24","arxiv_id":"2203.13319","n_code_links":0,"syntology":null},{"paper":"/paper/continual-sequence-generation-with-adaptive","slug":"continual-sequence-generation-with-adaptive","title":"Continual Sequence Generation with Adaptive Compositional Modules","date":"2022-03-20","arxiv_id":"2203.10652","n_code_links":2,"syntology":{"ran":14,"of":15,"n_ran_checked":14,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["GT-SALT/Adaptive-Compositional-Modules"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"9926b2a91703bd1960c384115c203d4b2b71d421944c3878416eb8350f51f29f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}