{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/experience-replay/papers/2","list_of":"/method/experience-replay","method":"Experience Replay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":9,"rows_per_page":100,"rows":[101,200],"of":865,"counts":{"archive_papers_tagged":865,"with_a_code_link":317,"where_syntology_ran_a_sample":94,"not_listed_spam_title":0,"listed":865,"listed_where_code_ran":94,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/experience-replay","prev":"/method/experience-replay","next":"/method/experience-replay/papers/3","papers":[{"paper":null,"slug":"cooperative-multi-agent-deep-reinforcement-4","title":"Cooperative Multi-Agent Deep Reinforcement Learning in Content Ranking Optimization","date":"2024-08-08","arxiv_id":"2408.04251","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-03088","title":"QADQN: Quantum Attention Deep Q-Network for Financial Market Prediction","date":"2024-08-06","arxiv_id":"2408.03088","n_code_links":0,"syntology":null},{"paper":null,"slug":"image-based-deep-reinforcement-learning-with","title":"Image-Based Deep Reinforcement Learning with Intrinsically Motivated Stimuli: On the Execution of Complex Robotic Tasks","date":"2024-07-31","arxiv_id":"2407.21338","n_code_links":0,"syntology":null},{"paper":"/paper/ftf-er-feature-topology-fusion-based","slug":"ftf-er-feature-topology-fusion-based","title":"FTF-ER: Feature-Topology Fusion-Based Experience Replay Method for Continual Graph Learning","date":"2024-07-28","arxiv_id":"2407.19429","n_code_links":1,"syntology":null},{"paper":null,"slug":"shangus-deep-reinforcement-learning-meets","title":"FH-DRL: Exponential-Hyperbolic Frontier Heuristics with DRL for accelerated Exploration in Unknown Environments","date":"2024-07-26","arxiv_id":"2407.18892","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-cross-environment-hyperparameter-setting","title":"The Cross-environment Hyperparameter Setting Benchmark for Reinforcement Learning","date":"2024-07-26","arxiv_id":"2407.18840","n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-to-memorize-and-to-forget-a-continual","title":"Learn to Memorize and to Forget: A Continual Learning Perspective of Dynamic SLAM","date":"2024-07-18","arxiv_id":"2407.13338","n_code_links":0,"syntology":null},{"paper":"/paper/reconfigurable-intelligent-surface-aided-21","slug":"reconfigurable-intelligent-surface-aided-21","title":"Reconfigurable Intelligent Surface Aided Vehicular Edge Computing: Joint Phase-shift Optimization and Multi-User Power Allocation","date":"2024-07-18","arxiv_id":"2407.13123","n_code_links":1,"syntology":null},{"paper":null,"slug":"er-fsl-experience-replay-with-feature","title":"ER-FSL: Experience Replay with Feature Subspace Learning for Online Continual Learning","date":"2024-07-17","arxiv_id":"2407.12279","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-with-symmetric-1","title":"Deep reinforcement learning with symmetric data augmentation applied for aircraft lateral attitude tracking control","date":"2024-07-13","arxiv_id":"2407.11077","n_code_links":0,"syntology":null},{"paper":"/paper/deep-attention-driven-reinforcement-learning","slug":"deep-attention-driven-reinforcement-learning","title":"Deep Attention Driven Reinforcement Learning (DAD-RL) for Autonomous Decision-Making in Dynamic Environment","date":"2024-07-12","arxiv_id":"2407.08932","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-the-interplay-of-prioritized","title":"Investigating the Interplay of Prioritized Replay and Generalization","date":"2024-07-12","arxiv_id":"2407.09702","n_code_links":0,"syntology":null},{"paper":null,"slug":"real-time-system-optimal-traffic-routing","title":"Real-time system optimal traffic routing under uncertainties -- Can physics models boost reinforcement learning?","date":"2024-07-10","arxiv_id":"2407.07364","n_code_links":0,"syntology":null},{"paper":"/paper/economic-span-selection-of-bridge-based-on","slug":"economic-span-selection-of-bridge-based-on","title":"Economic span selection of bridge based on deep reinforcement learning","date":"2024-07-09","arxiv_id":"2407.06507","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhanced-safety-in-autonomous-driving","title":"Enhanced Safety in Autonomous Driving: Integrating Latent State Diffusion Model for End-to-End Navigation","date":"2024-07-08","arxiv_id":"2407.06317","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-learning-optimizations-for-auto","title":"Continual Learning Optimizations for Auto-regressive Decoder of Multilingual ASR systems","date":"2024-07-04","arxiv_id":"2407.03645","n_code_links":0,"syntology":null},{"paper":"/paper/roer-regularized-optimal-experience-replay","slug":"roer-regularized-optimal-experience-replay","title":"ROER: Regularized Optimal Experience Replay","date":"2024-07-04","arxiv_id":"2407.03995","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-strategies-in","title":"Deep Reinforcement Learning Strategies in Finance: Insights into Asset Holding, Trading Behavior, and Purchase Diversity","date":"2024-06-29","arxiv_id":"2407.09557","n_code_links":0,"syntology":null},{"paper":null,"slug":"performance-comparison-of-deep-rl-algorithms-1","title":"Performance Comparison of Deep RL Algorithms for Mixed Traffic Cooperative Lane-Changing","date":"2024-06-25","arxiv_id":"2407.02521","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-universal-medical-image-registration","title":"Toward Universal Medical Image Registration via Sharpness-Aware Meta-Continual Learning","date":"2024-06-25","arxiv_id":"2406.17575","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-dynamic-resource-allocation-and","title":"Towards Dynamic Resource Allocation and Client Scheduling in Hierarchical Federated Learning: A Two-Phase Deep Reinforcement Learning Approach","date":"2024-06-21","arxiv_id":"2406.14910","n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-efficient-imitative-multi-token","title":"Physics-informed Imitative Reinforcement Learning for Real-world Driving","date":"2024-06-18","arxiv_id":"2407.02508","n_code_links":0,"syntology":null},{"paper":null,"slug":"cuer-corrected-uniform-experience-replay-for","title":"CUER: Corrected Uniform Experience Replay for Off-Policy Continuous Deep Reinforcement Learning Algorithms","date":"2024-06-13","arxiv_id":"2406.09030","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-opponent-policy-detection-in-multi","title":"Adaptive Opponent Policy Detection in Multi-Agent MDPs: Real-Time Strategy Switch Identification Using Running Error Estimation","date":"2024-06-10","arxiv_id":"2406.06500","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-continual-learning-of-video-diffusion","title":"Lifelong Learning of Video Diffusion Models From a Single Video Stream","date":"2024-06-07","arxiv_id":"2406.04814","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-view-on-planning-in-online","title":"A New View on Planning in Online Reinforcement Learning","date":"2024-06-03","arxiv_id":"2406.01562","n_code_links":0,"syntology":null},{"paper":null,"slug":"ikan-global-incremental-learning-with-kan-for","title":"iKAN: Global Incremental Learning with KAN for Human Activity Recognition Across Heterogeneous Datasets","date":"2024-06-03","arxiv_id":"2406.01646","n_code_links":0,"syntology":null},{"paper":null,"slug":"value-improved-actor-critic-algorithms","title":"Value Improved Actor Critic Algorithms","date":"2024-06-03","arxiv_id":"2406.01423","n_code_links":0,"syntology":null},{"paper":null,"slug":"shared-unique-features-and-task-aware","title":"Shared-unique Features and Task-aware Prioritized Sampling on Multi-task Reinforcement Learning","date":"2024-06-02","arxiv_id":"2406.00761","n_code_links":0,"syntology":null},{"paper":"/paper/saturn-sample-efficient-generative-molecular","slug":"saturn-sample-efficient-generative-molecular","title":"Saturn: Sample-efficient Generative Molecular Design using Memory Manipulation","date":"2024-05-27","arxiv_id":"2405.17066","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["schwallergroup/augmented_memory","schwallergroup/saturn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"meta-reinforcement-learning-for-resource-1","title":"Meta Reinforcement Learning for Resource Allocation in Multi-Antenna UAV Network with Rate Splitting Multiple Access","date":"2024-05-18","arxiv_id":"2405.11306","n_code_links":0,"syntology":null},{"paper":null,"slug":"chaos-based-reinforcement-learning-with-td3","title":"Chaos-based reinforcement learning with TD3","date":"2024-05-15","arxiv_id":"2405.09086","n_code_links":0,"syntology":null},{"paper":null,"slug":"mgser-sam-memory-guided-soft-experience","title":"MGSER-SAM: Memory-Guided Soft Experience Replay with Sharpness-Aware Optimization for Enhanced Continual Learning","date":"2024-05-15","arxiv_id":"2405.09492","n_code_links":0,"syntology":null},{"paper":null,"slug":"cier-a-novel-experience-replay-approach-with","title":"CIER: A Novel Experience Replay Approach with Causal Inference in Deep Reinforcement Learning","date":"2024-05-14","arxiv_id":"2405.08380","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-partial-survey-of-decentralized-cooperative","title":"An Initial Introduction to Cooperative Multi-Agent Reinforcement Learning","date":"2024-05-10","arxiv_id":"2405.06161","n_code_links":0,"syntology":null},{"paper":null,"slug":"offline-reinforcement-learning-with-8","title":"Offline Reinforcement Learning with Behavioral Supervisor Tuning","date":"2024-04-25","arxiv_id":"2404.16399","n_code_links":0,"syntology":null},{"paper":"/paper/a-fast-balance-optimization-approach-for","slug":"a-fast-balance-optimization-approach-for","title":"A fast balance optimization approach for charging enhancement of lithium-ion battery packs through deep reinforcement learning","date":"2024-04-24","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"single-task-continual-offline-reinforcement","title":"Data-Incremental Continual Offline Reinforcement Learning","date":"2024-04-19","arxiv_id":"2404.12639","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-policy-optimization-with-temporal-logic","title":"LTL-Constrained Policy Optimization with Cycle Experience Replay","date":"2024-04-17","arxiv_id":"2404.11578","n_code_links":0,"syntology":null},{"paper":null,"slug":"continuous-control-reinforcement-learning","title":"Continuous Control Reinforcement Learning: Distributed Distributional DrQ Algorithms","date":"2024-04-16","arxiv_id":"2404.10645","n_code_links":0,"syntology":null},{"paper":"/paper/joint-physical-digital-facial-attack","slug":"joint-physical-digital-facial-attack","title":"Joint Physical-Digital Facial Attack Detection Via Simulating Spoofing Clues","date":"2024-04-12","arxiv_id":"2404.08450","n_code_links":3,"syntology":null},{"paper":"/paper/continual-learning-with-weight-interpolation","slug":"continual-learning-with-weight-interpolation","title":"Continual Learning with Weight Interpolation","date":"2024-04-05","arxiv_id":"2404.04002","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jedrzejkozal/weight-interpolation-cl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/laser-learning-environment-a-new-environment","slug":"laser-learning-environment-a-new-environment","title":"Laser Learning Environment: A new environment for coordination-critical multi-agent tasks","date":"2024-04-04","arxiv_id":"2404.03596","n_code_links":1,"syntology":null},{"paper":null,"slug":"imitation-game-a-model-based-and-imitation","title":"Imitation Game: A Model-based and Imitation Learning Deep Reinforcement Learning Hybrid","date":"2024-04-02","arxiv_id":"2404.01794","n_code_links":0,"syntology":null},{"paper":null,"slug":"tuning-for-the-unknown-revisiting-evaluation","title":"K-percent Evaluation for Lifelong RL","date":"2024-04-02","arxiv_id":"2404.02113","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-improved-strategy-for-blood-glucose","title":"An Improved Strategy for Blood Glucose Control Using Multi-Step Deep Reinforcement Learning","date":"2024-03-12","arxiv_id":"2403.07566","n_code_links":0,"syntology":null},{"paper":null,"slug":"rlingua-improving-reinforcement-learning","title":"RLingua: Improving Reinforcement Learning Sample Efficiency in Robotic Manipulations With Large Language Models","date":"2024-03-11","arxiv_id":"2403.06420","n_code_links":0,"syntology":null},{"paper":null,"slug":"conservative-ddpg-pessimistic-rl-without","title":"Conservative DDPG -- Pessimistic RL without Ensemble","date":"2024-03-08","arxiv_id":"2403.05732","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-natural-extension-to-online-algorithms-for","title":"A Natural Extension To Online Algorithms For Hybrid RL With Limited Coverage","date":"2024-03-07","arxiv_id":"2403.09701","n_code_links":0,"syntology":null},{"paper":null,"slug":"fill-and-spill-deep-reinforcement-learning","title":"Fill-and-Spill: Deep Reinforcement Learning Policy Gradient Methods for Reservoir Operation Decision and Control","date":"2024-03-07","arxiv_id":"2403.04195","n_code_links":0,"syntology":null},{"paper":"/paper/llms-in-the-imaginarium-tool-learning-through","slug":"llms-in-the-imaginarium-tool-learning-through","title":"LLMs in the Imaginarium: Tool Learning through Simulated Trial and Error","date":"2024-03-07","arxiv_id":"2403.04746","n_code_links":1,"syntology":{"ran":6,"of":13,"n_ran_checked":6,"n_instrument":0,"unverified":7,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["microsoft/simulated-trial-and-error"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/analyzing-and-reducing-catastrophic","slug":"analyzing-and-reducing-catastrophic","title":"Analyzing and Reducing Catastrophic Forgetting in Parameter Efficient Tuning","date":"2024-02-29","arxiv_id":"2402.18865","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["which47/llmcl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"human-centric-aware-uav-trajectory-planning","title":"Human-Centric Aware UAV Trajectory Planning in Search and Rescue Missions Employing Multi-Objective Reinforcement Learning with AHP and Similarity-Based Experience Replay","date":"2024-02-28","arxiv_id":"2402.18487","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-free-deep-deterministic-policy-gradient","title":"Model Free Deep Deterministic Policy Gradient Controller for Setpoint Tracking of Non-minimum Phase Systems","date":"2024-02-27","arxiv_id":"2402.17703","n_code_links":0,"syntology":null},{"paper":"/paper/combinatorial-client-master-multiagent-deep","slug":"combinatorial-client-master-multiagent-deep","title":"Combinatorial Client-Master Multiagent Deep Reinforcement Learning for Task Offloading in Mobile Edge Computing","date":"2024-02-18","arxiv_id":"2402.11653","n_code_links":2,"syntology":null},{"paper":null,"slug":"reinforcement-learning-to-maximise-wind","title":"Reinforcement learning to maximise wind turbine energy generation","date":"2024-02-17","arxiv_id":"2402.11384","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploiting-estimation-bias-in-deep-double-q","title":"Exploiting Estimation Bias in Clipped Double Q-Learning for Continous Control Reinforcement Learning Tasks","date":"2024-02-14","arxiv_id":"2402.09078","n_code_links":0,"syntology":null},{"paper":"/paper/layerwise-proximal-replay-a-proximal-point","slug":"layerwise-proximal-replay-a-proximal-point","title":"Layerwise Proximal Replay: A Proximal Point Method for Online Continual Learning","date":"2024-02-14","arxiv_id":"2402.09542","n_code_links":1,"syntology":null},{"paper":"/paper/flowpg-action-constrained-policy-gradient-1","slug":"flowpg-action-constrained-policy-gradient-1","title":"FlowPG: Action-constrained Policy Gradient with Normalizing Flows","date":"2024-02-07","arxiv_id":"2402.05149","n_code_links":1,"syntology":null},{"paper":null,"slug":"textit-minmaxmin-q-learning","title":"MinMaxMin $Q$-learning","date":"2024-02-03","arxiv_id":"2402.05951","n_code_links":0,"syntology":null},{"paper":null,"slug":"textit-sqt-textit-std-q-target","title":"SQT -- std $Q$-target","date":"2024-02-03","arxiv_id":"2402.05950","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-style-transfer-with-twin-delayed-ddpg","title":"Neural Style Transfer with Twin-Delayed DDPG for Shared Control of Robotic Manipulators","date":"2024-02-01","arxiv_id":"2402.00722","n_code_links":0,"syntology":null},{"paper":"/paper/continuous-unsupervised-domain-adaptation","slug":"continuous-unsupervised-domain-adaptation","title":"Continuous Unsupervised Domain Adaptation Using Stabilized Representations and Experience Replay","date":"2024-01-31","arxiv_id":"2402.00580","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-voltage","title":"Deep Reinforcement Learning for Voltage Control and Renewable Accommodation Using Spatial-Temporal Graph Information","date":"2024-01-29","arxiv_id":"2401.15848","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-fast-changing-slow-in-spiking-neural","title":"Learning fast changing slow in spiking neural networks","date":"2024-01-25","arxiv_id":"2402.10069","n_code_links":0,"syntology":null},{"paper":null,"slug":"back-stepping-experience-replay-with","title":"Back-stepping Experience Replay with Application to Model-free Reinforcement Learning for a Soft Snake Robot","date":"2024-01-21","arxiv_id":"2401.11372","n_code_links":0,"syntology":null},{"paper":"/paper/reconciling-spatial-and-temporal-abstractions","slug":"reconciling-spatial-and-temporal-abstractions","title":"Reconciling Spatial and Temporal Abstractions for Goal Representation","date":"2024-01-18","arxiv_id":"2401.09870","n_code_links":1,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["cosynus-lix/STAR"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"solving-continual-offline-reinforcement","title":"Solving Continual Offline Reinforcement Learning with Decision Transformer","date":"2024-01-16","arxiv_id":"2401.08478","n_code_links":0,"syntology":null},{"paper":null,"slug":"sum-throughput-maximization-in-multi-bd","title":"Sum Throughput Maximization in Multi-BD Symbiotic Radio NOMA Network Assisted by Active-STAR-RIS","date":"2024-01-16","arxiv_id":"2401.08301","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-enabled-priority-and-auction-based","title":"AI-enabled Priority and Auction-Based Spectrum Management for 6G","date":"2024-01-12","arxiv_id":"2401.06484","n_code_links":0,"syntology":null},{"paper":"/paper/an-experimental-evaluation-of-deep","slug":"an-experimental-evaluation-of-deep","title":"An experimental evaluation of Deep Reinforcement Learning algorithms for HVAC control","date":"2024-01-11","arxiv_id":"2401.05737","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-topology-aware-graph-coarsening-framework","title":"A Topology-aware Graph Coarsening Framework for Continual Graph Learning","date":"2024-01-05","arxiv_id":"2401.03077","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-discounting-of-training-time-attacks","title":"Adaptive Discounting of Training Time Attacks","date":"2024-01-05","arxiv_id":"2401.02652","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-reinforcement-learning-for-local-path","title":"Deep Reinforcement Learning for Local Path Following of an Autonomous Formula SAE Vehicle","date":"2024-01-05","arxiv_id":"2401.02903","n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-guided-never-ending-learning-to","title":"Uncertainty-Guided Never-Ending Learning to Drive","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-kalman-based-hybrid-car-following","title":"Adaptive Kalman-based hybrid car following strategy using TD3 and CACC","date":"2023-12-26","arxiv_id":"2312.15993","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiagent-copilot-approach-for-shared","title":"Multiagent Copilot Approach for Shared Autonomy between Human EEG and TD3 Deep Reinforcement Learning","date":"2023-12-22","arxiv_id":"2312.14458","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-fairness-aware-spectrum-auction-for","title":"Dynamic Fairness-Aware Spectrum Auction for Enhanced Licensed Shared Access in 6G Networks","date":"2023-12-20","arxiv_id":"2312.12867","n_code_links":0,"syntology":null},{"paper":null,"slug":"contextual-reinforcement-learning-for","title":"Contextual Reinforcement Learning for Offshore Wind Farm Bidding","date":"2023-12-18","arxiv_id":"2312.10884","n_code_links":0,"syntology":null},{"paper":null,"slug":"class-wise-buffer-management-for-incremental","title":"Class-Wise Buffer Management for Incremental Object Detection: An Effective Buffer Training Strategy","date":"2023-12-14","arxiv_id":"2312.09139","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-robotic-navigation-an-evaluation-of","title":"Enhancing Robotic Navigation: An Evaluation of Single and Multi-Objective Reinforcement Learning Strategies","date":"2023-12-13","arxiv_id":"2312.07953","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-designing-multi-uav-aided-wireless-powered","title":"On Designing Multi-UAV aided Wireless Powered Dynamic Communication via Hierarchical Deep Reinforcement Learning","date":"2023-12-13","arxiv_id":"2312.07917","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-sparse-reward-goal-conditioned","slug":"efficient-sparse-reward-goal-conditioned","title":"Efficient Sparse-Reward Goal-Conditioned Reinforcement Learning with a High Replay Ratio and Regularization","date":"2023-12-10","arxiv_id":"2312.05787","n_code_links":1,"syntology":null},{"paper":null,"slug":"privacy-preserving-multi-agent-reinforcement","title":"Privacy Preserving Multi-Agent Reinforcement Learning in Supply Chains","date":"2023-12-09","arxiv_id":"2312.05686","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-horizon-reinforcement-learning-in","title":"Finite Horizon Multi-Agent Reinforcement Learning in Solving Optimal Control of State-Dependent Switched Systems","date":"2023-12-08","arxiv_id":"2312.04767","n_code_links":0,"syntology":null},{"paper":null,"slug":"contact-energy-based-hindsight-experience","title":"Contact Energy Based Hindsight Experience Prioritization","date":"2023-12-05","arxiv_id":"2312.02677","n_code_links":0,"syntology":null},{"paper":null,"slug":"directly-attention-loss-adjusted-prioritized","title":"Directly Attention Loss Adjusted Prioritized Experience Replay","date":"2023-11-24","arxiv_id":"2311.14390","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-timescale-control-and-communications-1","title":"Multi-Timescale Control and Communications with Deep Reinforcement Learning -- Part II: Control-Aware Radio Resource Allocation","date":"2023-11-19","arxiv_id":"2311.11280","n_code_links":0,"syntology":null},{"paper":null,"slug":"6g-fresnel-spot-beamfocusing-using-large","title":"6G Fresnel Spot Beamfocusing using Large-Scale Metasurfaces: A Distributed DRL-Based Approach","date":"2023-11-18","arxiv_id":"2311.11109","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-sensing-and-communication-optimization","title":"Joint Sensing and Communication Optimization in Target-Mounted STARS-Assisted Vehicular Networks: A MADRL Approach","date":"2023-11-17","arxiv_id":"2311.10352","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-algorithmic-trading-a-multi","title":"Advancing Algorithmic Trading: A Multi-Technique Enhancement of Deep Q-Network Models","date":"2023-11-09","arxiv_id":"2311.05743","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-learn-for-few-shot-continual","title":"Learning to Learn for Few-shot Continual Active Learning","date":"2023-11-07","arxiv_id":"2311.03732","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-estimation-errors-by-twin-td","title":"Mitigating Estimation Errors by Twin TD-Regularized Actor and Critic for Deep Reinforcement Learning","date":"2023-11-07","arxiv_id":"2311.03711","n_code_links":0,"syntology":null},{"paper":null,"slug":"epidemic-decision-making-system-based","title":"Epidemic Decision-making System Based Federated Reinforcement Learning","date":"2023-11-03","arxiv_id":"2311.01749","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoupled-actor-critic","title":"On the Theory of Risk-Aware Agents: Bridging Actor-Critic and Economics","date":"2023-10-30","arxiv_id":"2310.19527","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-v2x-autonomous-perception-from-road","title":"Dynamic V2X Autonomous Perception from Road-to-Vehicle Vision","date":"2023-10-29","arxiv_id":"2310.19113","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-framework-for-interpretable-and","slug":"hierarchical-framework-for-interpretable-and","title":"Hierarchical Framework for Interpretable and Probabilistic Model-Based Safe Reinforcement Learning","date":"2023-10-28","arxiv_id":"2310.18811","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-resource-management-for-edge-network","title":"Adaptive Resource Management for Edge Network Slicing using Incremental Multi-Agent Deep Reinforcement Learning","date":"2023-10-26","arxiv_id":"2310.17523","n_code_links":0,"syntology":null},{"paper":"/paper/model-predictive-control-based-value","slug":"model-predictive-control-based-value","title":"Model predictive control-based value estimation for efficient reinforcement learning","date":"2023-10-25","arxiv_id":"2310.16646","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-and-sample-complexity","title":"On the Convergence and Sample Complexity Analysis of Deep Q-Networks with $ε$-Greedy Exploration","date":"2023-10-24","arxiv_id":"2310.16173","n_code_links":0,"syntology":null}],"record_sha256":"51cdf49dd75e2b33a4bde2e2f4a686de25c74e55ae55dcc5ed0515d3b5ca1fea","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}