{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/3","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":9,"rows_per_page":100,"rows":[201,300],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay/papers/2","next":"/method/experience-replay/papers/4","papers":[{"paper":null,"slug":"using-experience-classification-for-training","title":"Using Experience Classification for Training Non-Markovian Tasks","date":"2023-10-18","arxiv_id":"2310.11678","n_code_links":0,"syntology":null},{"paper":null,"slug":"sim-to-real-transfer-of-adaptive-control","title":"Sim-to-Real Transfer of Adaptive Control Parameters for AUV Stabilization under Current Disturbance","date":"2023-10-17","arxiv_id":"2310.11075","n_code_links":0,"syntology":null},{"paper":"/paper/dsac-t-distributional-soft-actor-critic-with","slug":"dsac-t-distributional-soft-actor-critic-with","title":"Distributional Soft Actor-Critic with Three Refinements","date":"2023-10-09","arxiv_id":"2310.05858","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jingliang-duan/dsac-t","jingliang-duan/dsac-v2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"energy-management-in-a-cooperative-energy","title":"Energy Management in a Cooperative Energy Harvesting Wireless Sensor Network","date":"2023-10-09","arxiv_id":"2310.05911","n_code_links":0,"syntology":null},{"paper":"/paper/gear-a-gpu-centric-experience-replay-system","slug":"gear-a-gpu-centric-experience-replay-system","title":"GEAR: A GPU-Centric Experience Replay System for Large Reinforcement Learning Models","date":"2023-10-08","arxiv_id":"2310.05205","n_code_links":1,"syntology":null},{"paper":null,"slug":"class-incremental-learning-using-generative","title":"Class-Incremental Learning Using Generative Experience Replay Based on Time-aware Regularization","date":"2023-10-05","arxiv_id":"2310.03898","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-contrastive-spoken-language","title":"Continual Contrastive Spoken Language Understanding","date":"2023-10-04","arxiv_id":"2310.02699","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-for-15","title":"A Deep Reinforcement Learning Approach for Interactive Search with Sentence-level Feedback","date":"2023-10-03","arxiv_id":"2310.03043","n_code_links":0,"syntology":null},{"paper":"/paper/differentially-encoded-observation-spaces-for","slug":"differentially-encoded-observation-spaces-for","title":"Differentially Encoded Observation Spaces for Perceptive Reinforcement Learning","date":"2023-10-03","arxiv_id":"2310.01767","n_code_links":1,"syntology":null},{"paper":"/paper/learning-and-reusing-primitive-behaviours-to","slug":"learning-and-reusing-primitive-behaviours-to","title":"Learning and reusing primitive behaviours to improve Hindsight Experience Replay sample efficiency","date":"2023-10-03","arxiv_id":"2310.01827","n_code_links":1,"syntology":null},{"paper":null,"slug":"cad-clustering-and-deep-reinforcement","title":"CAD: Clustering And Deep Reinforcement Learning Based Multi-Period Portfolio Management Strategy","date":"2023-10-02","arxiv_id":"2310.01319","n_code_links":0,"syntology":null},{"paper":"/paper/cleanba-a-reproducible-and-efficient","slug":"cleanba-a-reproducible-and-efficient","title":"Cleanba: A Reproducible and Efficient Distributed Reinforcement Learning Platform","date":"2023-09-29","arxiv_id":"2310.00036","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["vwxyzjn/cleanba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/e2net-resource-efficient-continual-learning","slug":"e2net-resource-efficient-continual-learning","title":"E2Net: Resource-Efficient Continual Learning with Elastic Expansion Network","date":"2023-09-28","arxiv_id":"2309.16117","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-the-heat","title":"Deep Reinforcement Learning for the Heat Transfer Control of Pulsating Impinging Jets","date":"2023-09-25","arxiv_id":"2309.13955","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-data-efficiency-in-reinforcement","slug":"enhancing-data-efficiency-in-reinforcement","title":"Enhancing data efficiency in reinforcement learning: a novel imagination mechanism based on mesh information propagation","date":"2023-09-25","arxiv_id":"2309.14243","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-long-n-step-surrogate-stage-reward-for-deep","title":"A Long $N$-step Surrogate Stage Reward for Deep Reinforcement Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/belief-projection-based-reinforcement","slug":"belief-projection-based-reinforcement","title":"Belief Projection-Based Reinforcement Learning for Environments with Delayed Feedback","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"differ-decomposing-individual-reward-for-fair","title":"DIFFER:Decomposing Individual Reward for Fair Experience Replay in Multi-Agent Reinforcement Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/double-gumbel-q-learning","slug":"double-gumbel-q-learning","title":"Double Gumbel Q-Learning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-and-sample-complexity-1","title":"On the Convergence and Sample Complexity Analysis of Deep Q-Networks with $\\epsilon$-Greedy Exploration","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"prioritizing-samples-in-reinforcement","title":"Prioritizing Samples in Reinforcement Learning with Reducible Loss","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-hierarchical-reinforcement-learning-for","title":"Safe Hierarchical Reinforcement Learning for CubeSat Task Scheduling Based on Energy Consumption","date":"2023-09-21","arxiv_id":"2309.12004","n_code_links":0,"syntology":null},{"paper":null,"slug":"taylor-td-learning","title":"Taylor TD-learning","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-driven-patient-monitoring-with-multi-agent","title":"Adaptive Multi-Agent Deep Reinforcement Learning for Timely Healthcare Interventions","date":"2023-09-20","arxiv_id":"2309.10980","n_code_links":0,"syntology":null},{"paper":null,"slug":"distributional-estimation-of-data-uncertainty","title":"Distributional Estimation of Data Uncertainty for Surveillance Face Anti-spoofing","date":"2023-09-18","arxiv_id":"2309.09485","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-loss-adjusted-prioritized","title":"Attention Loss Adjusted Prioritized Experience Replay","date":"2023-09-13","arxiv_id":"2309.06684","n_code_links":0,"syntology":null},{"paper":null,"slug":"uer-a-heuristic-bias-addressing-approach-for","title":"UER: A Heuristic Bias Addressing Approach for Online Continual Learning","date":"2023-09-08","arxiv_id":"2309.04081","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-of-representation-learning-and","title":"Hybrid of representation learning and reinforcement learning for dynamic and complex robotic motion planning","date":"2023-09-07","arxiv_id":"2309.03758","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-safe-deep-reinforcement-learning-approach","title":"A Safe Deep Reinforcement Learning Approach for Energy Efficient Federated Learning in Wireless Communication Networks","date":"2023-08-21","arxiv_id":"2308.10664","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-fusion-of-variational-distribution-priors","title":"A Fusion of Variational Distribution Priors and Saliency Map Replay for Continual 3D Reconstruction","date":"2023-08-17","arxiv_id":"2308.08812","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaer-an-adaptive-experience-replay-approach","title":"AdaER: An Adaptive Experience Replay Approach for Continual Lifelong Learning","date":"2023-08-07","arxiv_id":"2308.03810","n_code_links":0,"syntology":null},{"paper":null,"slug":"vehicles-control-collision-avoidance-using","title":"Vehicles Control: Collision Avoidance using Federated Deep Reinforcement Learning","date":"2023-08-04","arxiv_id":"2308.02614","n_code_links":0,"syntology":null},{"paper":null,"slug":"caching-at-stars-the-next-generation-edge","title":"Caching-at-STARS: the Next Generation Edge Caching","date":"2023-08-01","arxiv_id":"2308.00562","n_code_links":0,"syntology":null},{"paper":null,"slug":"ether-aligning-emergent-communication-for","title":"ETHER: Aligning Emergent Communication for Hindsight Experience Replay","date":"2023-07-28","arxiv_id":"2307.15494","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-reinforcement-learning-techniques","slug":"exploring-reinforcement-learning-techniques","title":"Exploring reinforcement learning techniques for discrete and continuous control tasks in the MuJoCo environment","date":"2023-07-20","arxiv_id":"2307.11166","n_code_links":1,"syntology":null},{"paper":null,"slug":"replay-to-remember-continual-layer-specific","title":"Replay to Remember: Continual Layer-Specific Fine-tuning for German Speech Recognition","date":"2023-07-14","arxiv_id":"2307.07280","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-reinforcement-learning-for-strategic","title":"Safe Reinforcement Learning for Strategic Bidding of Virtual Power Plants in Day-Ahead Markets","date":"2023-07-11","arxiv_id":"2307.05812","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpretable-and-secure-trajectory","title":"Interpretable and Secure Trajectory Optimization for UAV-Assisted Communication","date":"2023-07-05","arxiv_id":"2307.02002","n_code_links":0,"syntology":null},{"paper":null,"slug":"safety-aware-task-composition-for-discrete","title":"Safety-Aware Task Composition for Discrete and Continuous Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.17033","n_code_links":0,"syntology":null},{"paper":null,"slug":"action-and-trajectory-planning-for-urban","title":"Action and Trajectory Planning for Urban Autonomous Driving with Hierarchical Reinforcement Learning","date":"2023-06-28","arxiv_id":"2306.15968","n_code_links":0,"syntology":null},{"paper":"/paper/curious-replay-for-model-based-adaptation","slug":"curious-replay-for-model-based-adaptation","title":"Curious Replay for Model-based Adaptation","date":"2023-06-28","arxiv_id":"2306.15934","n_code_links":1,"syntology":null},{"paper":"/paper/romo-her-robust-model-based-hindsight","slug":"romo-her-robust-model-based-hindsight","title":"MRHER: Model-based Relay Hindsight Experience Replay for Sequential Object Manipulation Tasks with Sparse Rewards","date":"2023-06-28","arxiv_id":"2306.16061","n_code_links":1,"syntology":null},{"paper":null,"slug":"evolutionary-strategy-guided-reinforcement","title":"Evolutionary Strategy Guided Reinforcement Learning via MultiBuffer Communication","date":"2023-06-20","arxiv_id":"2306.11535","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-efficient-comfort-and-energy-saving","title":"Safe, Efficient, Comfort, and Energy-saving Automated Driving through Roundabout Based on Deep Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11465","n_code_links":0,"syntology":null},{"paper":null,"slug":"vanishing-bias-heuristic-guided-reinforcement","title":"Vanishing Bias Heuristic-guided Reinforcement Learning Algorithm","date":"2023-06-17","arxiv_id":"2306.10216","n_code_links":0,"syntology":null},{"paper":null,"slug":"temporal-difference-learning-with-experience","title":"Temporal Difference Learning with Experience Replay","date":"2023-06-16","arxiv_id":"2306.09746","n_code_links":0,"syntology":null},{"paper":"/paper/offline-prioritized-experience-replay","slug":"offline-prioritized-experience-replay","title":"Decoupled Prioritized Resampling for Offline RL","date":"2023-06-08","arxiv_id":"2306.05412","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/oper","yueyang130/odpr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/progression-cognition-reinforcement-learning","slug":"progression-cognition-reinforcement-learning","title":"Progression Cognition Reinforcement Learning with Prioritized Experience for Multi-Vehicle Pursuit","date":"2023-06-08","arxiv_id":"2306.05016","n_code_links":1,"syntology":null},{"paper":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","n_code_links":1,"syntology":null},{"paper":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-environment-lifelong-deep-reinforcement","title":"Multi-environment lifelong deep reinforcement learning for medical imaging","date":"2023-05-31","arxiv_id":"2306.00188","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-promise-and-limits-of-real-time","slug":"exploring-the-promise-and-limits-of-real-time","title":"Exploring the Promise and Limits of Real-Time Recurrent Learning","date":"2023-05-30","arxiv_id":"2305.19044","n_code_links":1,"syntology":null},{"paper":null,"slug":"history-repeats-overcoming-catastrophic","title":"History Repeats: Overcoming Catastrophic Forgetting For Event-Centric Temporal Knowledge Graph Completion","date":"2023-05-30","arxiv_id":"2305.18675","n_code_links":0,"syntology":null},{"paper":"/paper/temporally-layered-architecture-for-efficient","slug":"temporally-layered-architecture-for-efficient","title":"Optimizing Attention and Cognitive Control Costs Using Temporally-Layered Architectures","date":"2023-05-30","arxiv_id":"2305.18701","n_code_links":1,"syntology":null},{"paper":null,"slug":"domo-ac-doubly-multi-step-off-policy-actor","title":"DoMo-AC: Doubly Multi-step Off-policy Actor-Critic Algorithm","date":"2023-05-29","arxiv_id":"2305.18501","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-with-reward-machines","title":"Reinforcement Learning With Reward Machines in Stochastic Games","date":"2023-05-27","arxiv_id":"2305.17372","n_code_links":0,"syntology":null},{"paper":"/paper/continual-learning-with-strong-experience","slug":"continual-learning-with-strong-experience","title":"Continual Learning with Strong Experience Replay","date":"2023-05-23","arxiv_id":"2305.13622","n_code_links":1,"syntology":null},{"paper":null,"slug":"offline-experience-replay-for-continual","title":"OER: Offline Experience Replay for Continual Offline Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.13804","n_code_links":0,"syntology":null},{"paper":"/paper/sharing-lifelong-reinforcement-learning","slug":"sharing-lifelong-reinforcement-learning","title":"Sharing Lifelong Reinforcement Learning Knowledge via Modulating Masks","date":"2023-05-18","arxiv_id":"2305.10997","n_code_links":2,"syntology":null},{"paper":null,"slug":"attention-based-qoe-aware-digital-twin","title":"Attention-based QoE-aware Digital Twin Empowered Edge Computing for Immersive Virtual Reality","date":"2023-05-15","arxiv_id":"2305.08569","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-practical-robust-reinforcement-learning","title":"On Practical Robust Reinforcement Learning: Practical Uncertainty Set and Double-Agent Algorithm","date":"2023-05-11","arxiv_id":"2305.06657","n_code_links":0,"syntology":null},{"paper":"/paper/extracting-diagnosis-pathways-from-electronic","slug":"extracting-diagnosis-pathways-from-electronic","title":"Extracting Diagnosis Pathways from Electronic Health Records Using Deep Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06295","n_code_links":1,"syntology":null},{"paper":"/paper/train-a-real-world-local-path-planner-in-one","slug":"train-a-real-world-local-path-planner-in-one","title":"Train a Real-world Local Path Planner in One Hour via Partially Decoupled Reinforcement Learning and Vectorized Diversity","date":"2023-05-07","arxiv_id":"2305.04180","n_code_links":1,"syntology":null},{"paper":"/paper/simple-noisy-environment-augmentation-for","slug":"simple-noisy-environment-augmentation-for","title":"Simple Noisy Environment Augmentation for Reinforcement Learning","date":"2023-05-04","arxiv_id":"2305.02882","n_code_links":1,"syntology":null},{"paper":"/paper/map-based-experience-replay-a-memory","slug":"map-based-experience-replay-a-memory","title":"Map-based Experience Replay: A Memory-Efficient Solution to Catastrophic Forgetting in Reinforcement Learning","date":"2023-05-03","arxiv_id":"2305.02054","n_code_links":1,"syntology":null},{"paper":"/paper/replay-memory-as-an-empirical-mdp-combining","slug":"replay-memory-as-an-empirical-mdp-combining","title":"Replay Memory as An Empirical MDP: Combining Conservative Estimation with Experience Replay","date":"2023-05-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"federated-deep-reinforcement-learning-for-thz","title":"Federated Deep Reinforcement Learning for THz-Beam Search with Limited CSI","date":"2023-04-25","arxiv_id":"2304.13109","n_code_links":0,"syntology":null},{"paper":"/paper/regularizing-second-order-influences-for","slug":"regularizing-second-order-influences-for","title":"Regularizing Second-Order Influences for Continual Learning","date":"2023-04-20","arxiv_id":"2304.10177","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["feifeiobama/InfluenceCL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"quantum-deep-q-learning-with-distributed","title":"Quantum deep Q learning with distributed prioritized experience replay","date":"2023-04-19","arxiv_id":"2304.09648","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-ris-aided-eh-noma-networks-a-deep","title":"Active RIS-aided EH-NOMA Networks: A Deep Reinforcement Learning Approach","date":"2023-04-11","arxiv_id":"2304.12184","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-reinforcement-learning","title":"Understanding Reinforcement Learning Algorithms: The Progress from Basic Q-learning to Proximal Policy Optimization","date":"2023-03-31","arxiv_id":"2304.00026","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-continual-learning-of-diffusion","title":"Exploring Continual Learning of Diffusion Models","date":"2023-03-27","arxiv_id":"2303.15342","n_code_links":0,"syntology":null},{"paper":"/paper/preserving-linear-separability-in-continual","slug":"preserving-linear-separability-in-continual","title":"Preserving Linear Separability in Continual Learning by Backward Feature Projection","date":"2023-03-26","arxiv_id":"2303.14595","n_code_links":1,"syntology":null},{"paper":"/paper/assessor-guided-learning-for-continual","slug":"assessor-guided-learning-for-continual","title":"Assessor-Guided Learning for Continual Environments","date":"2023-03-21","arxiv_id":"2303.11624","n_code_links":1,"syntology":null},{"paper":null,"slug":"energy-management-of-multi-mode-plug-in","title":"Energy Management of Multi-mode Plug-in Hybrid Electric Vehicle using Multi-agent Deep Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09658","n_code_links":0,"syntology":null},{"paper":null,"slug":"svde-scalable-value-decomposition-exploration","title":"SVDE: Scalable Value-Decomposition Exploration for Cooperative Multi-Agent Reinforcement Learning","date":"2023-03-16","arxiv_id":"2303.09058","n_code_links":0,"syntology":null},{"paper":"/paper/synthetic-experience-replay-1","slug":"synthetic-experience-replay-1","title":"Synthetic Experience Replay","date":"2023-03-12","arxiv_id":"2303.06614","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["conglu1997/SynthER"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":null,"slug":"a-strategy-oriented-bayesian-soft-actor","title":"A Strategy-Oriented Bayesian Soft Actor-Critic Model","date":"2023-03-07","arxiv_id":"2303.04193","n_code_links":0,"syntology":null},{"paper":"/paper/swim-a-general-purpose-high-performing-and","slug":"swim-a-general-purpose-high-performing-and","title":"Swim: A General-Purpose, High-Performing, and Efficient Activation Function for Locomotion Control Tasks","date":"2023-03-05","arxiv_id":"2303.02640","n_code_links":1,"syntology":null},{"paper":null,"slug":"eventual-discounting-temporal-logic","title":"Eventual Discounting Temporal Logic Counterfactual Experience Replay","date":"2023-03-03","arxiv_id":"2303.02135","n_code_links":0,"syntology":null},{"paper":null,"slug":"hindsight-states-blending-sim-and-real-task","title":"Hindsight States: Blending Sim and Real Task Elements for Efficient Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.02234","n_code_links":0,"syntology":null},{"paper":"/paper/can-bert-refrain-from-forgetting-on","slug":"can-bert-refrain-from-forgetting-on","title":"Can BERT Refrain from Forgetting on Sequential Tasks? A Probing Study","date":"2023-03-02","arxiv_id":"2303.01081","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-ladder-in-chaos-a-simple-and-effective","title":"The Ladder in Chaos: A Simple and Effective Improvement to General DRL Algorithms by Policy Path Trimming and Boosting","date":"2023-03-02","arxiv_id":"2303.01391","n_code_links":0,"syntology":null},{"paper":null,"slug":"combating-uncertainties-in-wind-and","title":"Combating Uncertainties in Wind and Distributed PV Energy Sources Using Integrated Reinforcement Learning and Time-Series Forecasting","date":"2023-02-27","arxiv_id":"2302.14094","n_code_links":0,"syntology":null},{"paper":null,"slug":"detachedly-learn-a-classifier-for-class","title":"Detachedly Learn a Classifier for Class-Incremental Learning","date":"2023-02-23","arxiv_id":"2302.11730","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-the-gumbel-softmax-in-maddpg","slug":"revisiting-the-gumbel-softmax-in-maddpg","title":"Revisiting the Gumbel-Softmax in MADDPG","date":"2023-02-23","arxiv_id":"2302.11793","n_code_links":1,"syntology":null},{"paper":null,"slug":"selective-experience-replay-compression-using","title":"Selective experience replay compression using coresets for lifelong deep reinforcement learning in medical imaging","date":"2023-02-22","arxiv_id":"2302.11510","n_code_links":0,"syntology":null},{"paper":"/paper/mac-po-multi-agent-experience-replay-via","slug":"mac-po-multi-agent-experience-replay-via","title":"MAC-PO: Multi-Agent Experience Replay via Collective Priority Optimization","date":"2023-02-21","arxiv_id":"2302.10418","n_code_links":1,"syntology":null},{"paper":null,"slug":"uav-path-planning-employing-mpc-reinforcement","title":"UAV Path Planning Employing MPC- Reinforcement Learning Method Considering Collision Avoidance","date":"2023-02-21","arxiv_id":"2302.10669","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-the-effect-of-varying-amounts","title":"Understanding the effect of varying amounts of replay per step","date":"2023-02-20","arxiv_id":"2302.10311","n_code_links":0,"syntology":null},{"paper":null,"slug":"prioritized-offline-goal-swapping-experience","title":"Prioritized offline Goal-swapping Experience Replay","date":"2023-02-15","arxiv_id":"2302.07741","n_code_links":0,"syntology":null},{"paper":"/paper/smart-6g-sky-for-green-mobile-iot-networks","slug":"smart-6g-sky-for-green-mobile-iot-networks","title":"Smart 6G Sky for Green Mobile IOT Networks","date":"2023-02-15","arxiv_id":"2302.09022","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-song-of-ice-and-fire-analyzing-textual","title":"A Song of Ice and Fire: Analyzing Textual Autotelic Agents in ScienceWorld","date":"2023-02-10","arxiv_id":"2302.05244","n_code_links":0,"syntology":null},{"paper":null,"slug":"ris-assisted-jamming-rejection-and-path","title":"RIS-Assisted Jamming Rejection and Path Planning for UAV-Borne IoT Platform: A New Deep Reinforcement Learning Framework","date":"2023-02-10","arxiv_id":"2302.04994","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-robust-inductive-graph-incremental","title":"On the Limitation and Experience Replay for GNNs in Continual Learning","date":"2023-02-07","arxiv_id":"2302.03534","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-traffic-light-1","title":"Deep Reinforcement Learning for Traffic Light Control in Intelligent Transportation Systems","date":"2023-02-04","arxiv_id":"2302.03669","n_code_links":0,"syntology":null},{"paper":null,"slug":"v2n-service-scaling-with-deep-reinforcement","title":"V2N Service Scaling with Deep Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.13324","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-internet-scale-vision-language","title":"Distilling Internet-Scale Vision-Language Models into Embodied Agents","date":"2023-01-29","arxiv_id":"2301.12507","n_code_links":0,"syntology":null}],"record_sha256":"dae8dd7713ed090c44a05e8efa178568c505b3079468782e9779feeee47d43c6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}