{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/18","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":59,"rows_per_page":100,"rows":[1701,1800],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/17","next":"/task/deep-reinforcement-learning/papers/19","papers":[{"url":"/paper/darla-improving-zero-shot-transfer-in","slug":"darla-improving-zero-shot-transfer-in","title":"DARLA: Improving Zero-Shot Transfer in Reinforcement Learning","date":"2017-07-26","arxiv_id":"1707.08475","repositories_listed":1,"syntology":null},{"url":"/paper/lenient-multi-agent-deep-reinforcement","slug":"lenient-multi-agent-deep-reinforcement","title":"Lenient Multi-Agent Deep Reinforcement Learning","date":"2017-07-14","arxiv_id":"1707.04402","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-from-observation-learning-to","slug":"imitation-from-observation-learning-to","title":"Imitation from Observation: Learning to Imitate Behaviors from Raw Video via Context Translation","date":"2017-07-11","arxiv_id":"1707.03374","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-behaviors-from-motion-capture","slug":"learning-human-behaviors-from-motion-capture","title":"Learning human behaviors from motion capture by adversarial imitation","date":"2017-07-07","arxiv_id":"1707.02201","repositories_listed":1,"syntology":null},{"url":"/paper/maintaining-cooperation-in-complex-social","slug":"maintaining-cooperation-in-complex-social","title":"Maintaining cooperation in complex social dilemmas using deep reinforcement learning","date":"2017-07-04","arxiv_id":"1707.01068","repositories_listed":1,"syntology":null},{"url":"/paper/action-decision-networks-for-visual-tracking","slug":"action-decision-networks-for-visual-tracking","title":"Action-Decision Networks for Visual Tracking With Deep Reinforcement Learning","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dex-incremental-learning-for-complex","slug":"dex-incremental-learning-for-complex","title":"Dex: Incremental Learning for Complex Environments in Deep Reinforcement Learning","date":"2017-06-19","arxiv_id":"1706.05749","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-task-generalization-with-multi-task","slug":"zero-shot-task-generalization-with-multi-task","title":"Zero-Shot Task Generalization with Multi-Task Deep Reinforcement Learning","date":"2017-06-15","arxiv_id":"1706.05064","repositories_listed":1,"syntology":null},{"url":"/paper/on-improving-deep-reinforcement-learning-for","slug":"on-improving-deep-reinforcement-learning-for","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2017-04-26","arxiv_id":"1704.07978","repositories_listed":1,"syntology":null},{"url":"/paper/modular-multi-objective-deep-reinforcement","slug":"modular-multi-objective-deep-reinforcement","title":"Modular Multi-Objective Deep Reinforcement Learning with Decision Values","date":"2017-04-21","arxiv_id":"1704.06676","repositories_listed":1,"syntology":null},{"url":"/paper/beating-atari-with-natural-language-guided","slug":"beating-atari-with-natural-language-guided","title":"Beating Atari with Natural Language Guided Reinforcement Learning","date":"2017-04-18","arxiv_id":"1704.05539","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-framework-for","slug":"deep-reinforcement-learning-framework-for","title":"Deep Reinforcement Learning framework for Autonomous Driving","date":"2017-04-08","arxiv_id":"1704.02532","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-simplification-with-deep","slug":"sentence-simplification-with-deep","title":"Sentence Simplification with Deep Reinforcement Learning","date":"2017-03-31","arxiv_id":"1703.10931","repositories_listed":1,"syntology":null},{"url":"/paper/ex2-exploration-with-exemplar-models-for-deep","slug":"ex2-exploration-with-exemplar-models-for-deep","title":"EX2: Exploration with Exemplar Models for Deep Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01260","repositories_listed":1,"syntology":null},{"url":"/paper/neural-map-structured-memory-for-deep","slug":"neural-map-structured-memory-for-deep","title":"Neural Map: Structured Memory for Deep Reinforcement Learning","date":"2017-02-27","arxiv_id":"1702.08360","repositories_listed":1,"syntology":null},{"url":"/paper/beating-the-worlds-best-at-super-smash-bros","slug":"beating-the-worlds-best-at-super-smash-bros","title":"Beating the World's Best at Super Smash Bros. with Deep Reinforcement Learning","date":"2017-02-21","arxiv_id":"1702.06230","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-visual-tracking-by-deep-reinforced","slug":"real-time-visual-tracking-by-deep-reinforced","title":"Real-time visual tracking by deep reinforced decision making","date":"2017-02-21","arxiv_id":"1702.06291","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-deep-reinforcement-learning","slug":"collaborative-deep-reinforcement-learning","title":"Collaborative Deep Reinforcement Learning","date":"2017-02-19","arxiv_id":"1702.05796","repositories_listed":1,"syntology":null},{"url":"/paper/vulnerability-of-deep-reinforcement-learning","slug":"vulnerability-of-deep-reinforcement-learning","title":"Vulnerability of Deep Reinforcement Learning to Policy Induction Attacks","date":"2017-01-16","arxiv_id":"1701.04143","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-via-recurrent","slug":"reinforcement-learning-via-recurrent","title":"Reinforcement Learning via Recurrent Convolutional Neural Networks","date":"2017-01-09","arxiv_id":"1701.02392","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-deep-network-solutions-for","slug":"a-survey-of-deep-network-solutions-for","title":"A Survey of Deep Network Solutions for Learning Control in Robotics: From Reinforcement to Imitation","date":"2016-12-21","arxiv_id":"1612.07139","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-optimization-with-robust-bayesian","slug":"bayesian-optimization-with-robust-bayesian","title":"Bayesian Optimization with Robust Bayesian Neural Networks","date":"2016-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/playing-doom-with-slam-augmented-deep","slug":"playing-doom-with-slam-augmented-deep","title":"Playing Doom with SLAM-Augmented Deep Reinforcement Learning","date":"2016-12-01","arxiv_id":"1612.00380","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-domain","slug":"deep-reinforcement-learning-for-multi-domain","title":"Deep Reinforcement Learning for Multi-Domain Dialogue Systems","date":"2016-11-26","arxiv_id":"1611.08675","repositories_listed":1,"syntology":null},{"url":"/paper/training-an-interactive-humanoid-robot-using","slug":"training-an-interactive-humanoid-robot-using","title":"Training an Interactive Humanoid Robot Using Multimodal Deep Reinforcement Learning","date":"2016-11-26","arxiv_id":"1611.08666","repositories_listed":1,"syntology":null},{"url":"/paper/cad2rl-real-single-image-flight-without-a","slug":"cad2rl-real-single-image-flight-without-a","title":"CAD2RL: Real Single-Image Flight without a Single Real Image","date":"2016-11-13","arxiv_id":"1611.04201","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-object-detection-with-deep","slug":"hierarchical-object-detection-with-deep","title":"Hierarchical Object Detection with Deep Reinforcement Learning","date":"2016-11-11","arxiv_id":"1611.03718","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-in-a-day-faster-deep","slug":"learning-to-play-in-a-day-faster-deep","title":"Learning to Play in a Day: Faster Deep Reinforcement Learning by Optimality Tightening","date":"2016-11-05","arxiv_id":"1611.01606","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-mention","slug":"deep-reinforcement-learning-for-mention","title":"Deep Reinforcement Learning for Mention-Ranking Coreference Models","date":"2016-09-27","arxiv_id":"1609.08667","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-mention#ran","syntology_url":"https://syntology.ai/paper/1609.08667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.08667"}},"official":{"repos":["clarkkev/deep-coref"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/opponent-modeling-in-deep-reinforcement","slug":"opponent-modeling-in-deep-reinforcement","title":"Opponent Modeling in Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05559","repositories_listed":1,"syntology":null},{"url":"/paper/playing-atari-games-with-deep-reinforcement","slug":"playing-atari-games-with-deep-reinforcement","title":"Playing Atari Games with Deep Reinforcement Learning and Human Checkpoint Replay","date":"2016-07-18","arxiv_id":"1607.05077","repositories_listed":1,"syntology":null},{"url":"/paper/actor-critic-versus-direct-policy-search-a","slug":"actor-critic-versus-direct-policy-search-a","title":"Actor-critic versus direct policy search: a comparison based on sample complexity","date":"2016-06-29","arxiv_id":"1606.09152","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-a","slug":"deep-reinforcement-learning-with-a","title":"Deep Reinforcement Learning with a Combinatorial Action Space for Predicting Popular Reddit Threads","date":"2016-06-12","arxiv_id":"1606.03667","repositories_listed":1,"syntology":null},{"url":"/paper/deep-successor-reinforcement-learning","slug":"deep-successor-reinforcement-learning","title":"Deep Successor Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02396","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-successor-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1606.02396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.02396"}},"official":{"repos":["Ardavans/DSR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-end-to-end-learning-for-dialog-state","slug":"towards-end-to-end-learning-for-dialog-state","title":"Towards End-to-End Learning for Dialog State Tracking and Management using Deep Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02560","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-radio-control-and","slug":"deep-reinforcement-learning-radio-control-and","title":"Deep Reinforcement Learning Radio Control and Signal Detection with KeRLym, a Gym RL Agent","date":"2016-05-30","arxiv_id":"1605.09221","repositories_listed":1,"syntology":null},{"url":"/paper/simpleds-a-simple-deep-reinforcement-learning","slug":"simpleds-a-simple-deep-reinforcement-learning","title":"SimpleDS: A Simple Deep Reinforcement Learning Dialogue System","date":"2016-01-18","arxiv_id":"1601.04574","repositories_listed":1,"syntology":null},{"url":"/paper/strategic-dialogue-management-via-deep","slug":"strategic-dialogue-management-via-deep","title":"Strategic Dialogue Management via Deep Reinforcement Learning","date":"2015-11-25","arxiv_id":"1511.08099","repositories_listed":1,"syntology":null},{"url":"/paper/policy-distillation","slug":"policy-distillation","title":"Policy Distillation","date":"2015-11-19","arxiv_id":"1511.06295","repositories_listed":1,"syntology":null},{"url":null,"slug":"lilm-rdb-sfc-lightweight-language-model-with","title":"LiLM-RDB-SFC: Lightweight Language Model with Relational Database-Guided DRL for Optimized SFC Provisioning","date":"2025-07-15","arxiv_id":"2507.10903","repositories_listed":0,"syntology":null},{"url":null,"slug":"sensing-accuracy-optimization-for-multi-uav","title":"Sensing Accuracy Optimization for Multi-UAV SAR Interferometry with Data Offloading","date":"2025-07-15","arxiv_id":"2507.11284","repositories_listed":0,"syntology":null},{"url":null,"slug":"turning-sand-to-gold-recycling-data-to-bridge","title":"Turning Sand to Gold: Recycling Data to Bridge On-Policy and Off-Policy Learning via Causal Bound","date":"2025-07-15","arxiv_id":"2507.11269","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-fast-and-data","title":"Meta-Reinforcement Learning for Fast and Data-Efficient Spectrum Allocation in Dynamic Wireless Networks","date":"2025-07-13","arxiv_id":"2507.10619","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-task-offloading-for-uav-assisted","title":"Hierarchical Task Offloading for UAV-Assisted Vehicular Edge Computing via Deep Reinforcement Learning","date":"2025-07-08","arxiv_id":"2507.05722","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-training-time-poisoning-component","title":"Beyond Training-time Poisoning: Component-level and Post-training Backdoors in Deep Reinforcement Learning","date":"2025-07-07","arxiv_id":"2507.04883","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-ai-for-radar-resource-management","title":"Explainable AI for Radar Resource Management: Modified LIME in Deep Reinforcement Learning","date":"2025-06-26","arxiv_id":"2506.20916","repositories_listed":0,"syntology":null},{"url":null,"slug":"rqdia-regularizing-q-value-distributions-with-1","title":"rQdia: Regularizing Q-Value Distributions With Image Augmentation","date":"2025-06-26","arxiv_id":"2506.21367","repositories_listed":0,"syntology":null},{"url":null,"slug":"gympn-a-library-for-decision-making-in","title":"GymPN: A Library for Decision-Making in Process Management Systems","date":"2025-06-25","arxiv_id":"2506.20404","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-resource-management-in","title":"Learning-Based Resource Management in Integrated Sensing and Communication Systems","date":"2025-06-25","arxiv_id":"2506.20849","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-reinforcement-learning-for-6","title":"Multi-Objective Reinforcement Learning for Cognitive Radar Resource Management","date":"2025-06-25","arxiv_id":"2506.20853","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-beam-selection-for-isac-in-cell","title":"Efficient Beam Selection for ISAC in Cell-Free Massive MIMO via Digital Twin-Assisted Deep Reinforcement Learning","date":"2025-06-23","arxiv_id":"2506.18560","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-design-of-experiment-for","title":"Optimal Design of Experiment for Electrochemical Parameter Identification of Li-ion Battery via Deep Reinforcement Learning","date":"2025-06-23","arxiv_id":"2506.19146","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-social-metaverse-streaming-based-on","title":"Adaptive Social Metaverse Streaming based on Federated Multi-Agent Deep Reinforcement Learning","date":"2025-06-19","arxiv_id":"2506.17342","repositories_listed":0,"syntology":null},{"url":null,"slug":"bida-a-bi-level-interaction-decision-making","title":"BIDA: A Bi-level Interaction Decision-making Algorithm for Autonomous Vehicles in Dynamic Traffic Scenarios","date":"2025-06-19","arxiv_id":"2506.16546","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-vidar-device-with-visual-inertial","title":"A Novel ViDAR Device With Visual Inertial Encoder Odometry and Reinforcement Learning-Based Active SLAM Method","date":"2025-06-16","arxiv_id":"2506.13100","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-spectrum-sensing-and-resource-1","title":"Joint Spectrum Sensing and Resource Allocation for OFDMA-based Underwater Acoustic Communications","date":"2025-06-16","arxiv_id":"2506.13008","repositories_listed":0,"syntology":null},{"url":"/paper/the-courage-to-stop-overcoming-sunk-cost","slug":"the-courage-to-stop-overcoming-sunk-cost","title":"The Courage to Stop: Overcoming Sunk Cost Fallacy in Deep Reinforcement Learning","date":"2025-06-16","arxiv_id":"2506.13672","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-courage-to-stop-overcoming-sunk-cost#ran","syntology_url":"https://syntology.ai/paper/2506.13672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13672"}},"official":null}},{"url":null,"slug":"federated-neuroevolution-o-ran-enhancing-the","title":"Federated Neuroevolution O-RAN: Enhancing the Robustness of Deep Reinforcement Learning xApps","date":"2025-06-15","arxiv_id":"2506.12812","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-treatment-planning-for-interstitial","title":"Automated Treatment Planning for Interstitial HDR Brachytherapy for Locally Advanced Cervical Cancer using Deep Reinforcement Learning","date":"2025-06-13","arxiv_id":"2506.11957","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-beamforming-with-extremely-large-scale","title":"Joint Beamforming with Extremely Large Scale RIS: A Sequential Multi-Agent A2C Approach","date":"2025-06-12","arxiv_id":"2506.10815","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-09562","title":"TooBadRL: Trigger Optimization to Boost Effectiveness of Backdoor Attacks on Deep Reinforcement Learning","date":"2025-06-11","arxiv_id":"2506.09562","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10073","title":"Patient-Specific Deep Reinforcement Learning for Automatic Replanning in Head-and-Neck Cancer Proton Therapy","date":"2025-06-11","arxiv_id":"2506.10073","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundation-model-aided-deep-reinforcement","title":"Foundation Model-Aided Deep Reinforcement Learning for RIS-Assisted Wireless Communication","date":"2025-06-11","arxiv_id":"2506.09855","repositories_listed":0,"syntology":null},{"url":null,"slug":"moorl-a-framework-for-integrating-offline","title":"MOORL: A Framework for Integrating Offline-Online Reinforcement Learning","date":"2025-06-11","arxiv_id":"2506.09574","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergizing-reinforcement-learning-and","title":"Synergizing Reinforcement Learning and Genetic Algorithms for Neural Combinatorial Optimization","date":"2025-06-11","arxiv_id":"2506.09404","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08630","title":"Modular Recurrence in Contextual MDPs for Universal Morphology Control","date":"2025-06-10","arxiv_id":"2506.08630","repositories_listed":0,"syntology":null},{"url":null,"slug":"preference-driven-multi-objective","title":"Preference-Driven Multi-Objective Combinatorial Optimization with Conditional Computation","date":"2025-06-10","arxiv_id":"2506.08898","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-economical-uav-trajectory-planning","title":"Safe and Economical UAV Trajectory Planning in Low-Altitude Airspace: A Hybrid DRL-LLM Approach with Compliance Awareness","date":"2025-06-10","arxiv_id":"2506.08532","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-deep-reinforcement-learning-1","title":"Towards Robust Deep Reinforcement Learning against Environmental State Perturbation","date":"2025-06-10","arxiv_id":"2506.08961","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08192","title":"Interpreting Agent Behaviors in Reinforcement-Learning-Based Cyber-Battle Simulation Platforms","date":"2025-06-09","arxiv_id":"2506.08192","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-intelligent-fault-self-healing-mechanism","title":"An Intelligent Fault Self-Healing Mechanism for Cloud AI Systems via Integration of Large Language Models and Deep Reinforcement Learning","date":"2025-06-09","arxiv_id":"2506.07411","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-near","title":"Deep reinforcement learning for near-deterministic preparation of cubic- and quartic-phase gates in photonic quantum computing","date":"2025-06-09","arxiv_id":"2506.07859","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-joint-real","title":"Deep reinforcement learning-based joint real-time energy scheduling for green buildings with heterogeneous battery energy storage devices","date":"2025-06-07","arxiv_id":"2506.06824","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-deep-reinforcement-learning-1","title":"Energy-efficient Deep Reinforcement Learning-based Network Function Disaggregation in Hybrid Non-terrestrial Open Radio Access Networks","date":"2025-06-07","arxiv_id":"2506.06876","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-choice-model-specification-using","title":"Improving choice model specification using reinforcement learning","date":"2025-06-06","arxiv_id":"2506.06410","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-economic-dispatch-of-power-to-gas-systems","title":"The Economic Dispatch of Power-to-Gas Systems with Deep Reinforcement Learning:Tackling the Challenge of Delayed Rewards with Long-Term Energy Storage","date":"2025-06-06","arxiv_id":"2506.06484","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-artificial-intelligence-trade-the-stock","title":"Can Artificial Intelligence Trade the Stock Market?","date":"2025-06-05","arxiv_id":"2506.04658","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-task-oriented-dialogue-policy","title":"An Efficient Task-Oriented Dialogue Policy: Evolutionary Reinforcement Learning Injected by Elite Individuals","date":"2025-06-04","arxiv_id":"2506.03519","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-vehicle-lateral-control-using-deep","title":"Autonomous Vehicle Lateral Control Using Deep Reinforcement Learning with MPC-PID Demonstration","date":"2025-06-04","arxiv_id":"2506.04040","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-beamforming-and-resource-allocation-for","title":"Beamforming and Resource Allocation for Delay Optimization in RIS-Assisted OFDM Systems","date":"2025-06-04","arxiv_id":"2506.03586","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-deep-reinforcement-learning-method","title":"A Novel Deep Reinforcement Learning Method for Computation Offloading in Multi-User Mobile Edge Computing with Decentralization","date":"2025-06-03","arxiv_id":"2506.02458","repositories_listed":0,"syntology":null},{"url":null,"slug":"eden-entorhinal-driven-egocentric-navigation","title":"EDEN: Entorhinal Driven Egocentric Navigation Toward Robotic Deployment","date":"2025-06-03","arxiv_id":"2506.03046","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximizing-the-promptness-of-metaverse","title":"Maximizing the Promptness of Metaverse Systems using Edge Computing by Deep Reinforcement Learning","date":"2025-06-03","arxiv_id":"2506.02657","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-pod-repositioning-problem-with","title":"Solving the Pod Repositioning Problem with Deep Reinforced Adaptive Large Neighborhood Search","date":"2025-06-03","arxiv_id":"2506.02746","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-reinforcement-learning-for-heat","title":"Interpretable reinforcement learning for heat pump control through asymmetric differentiable decision trees","date":"2025-06-02","arxiv_id":"2506.01641","repositories_listed":0,"syntology":null},{"url":null,"slug":"basil-best-action-symbolic-interpretable","title":"BASIL: Best-Action Symbolic Interpretable Learning for Evolving Compact RL Policies","date":"2025-05-31","arxiv_id":"2506.00328","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-hanabi","title":"Reinforcement Learning for Hanabi","date":"2025-05-31","arxiv_id":"2506.00458","repositories_listed":0,"syntology":null},{"url":null,"slug":"axiom-learning-to-play-games-in-minutes-with","title":"AXIOM: Learning to Play Games in Minutes with Expanding Object-Centric Models","date":"2025-05-30","arxiv_id":"2505.24784","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-sensory-musculoskeletal-modeling-and","title":"Human sensory-musculoskeletal modeling and control of whole-body movements","date":"2025-05-29","arxiv_id":"2506.00071","repositories_listed":0,"syntology":null},{"url":null,"slug":"measure-gradients-not-activations-enhancing","title":"Measure gradients, not activations! Enhancing neuronal activity in deep reinforcement learning","date":"2025-05-29","arxiv_id":"2505.24061","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-enhanced-prompt-decision","title":"Attention-Enhanced Prompt Decision Transformers for UAV-Assisted Communications with AoI","date":"2025-05-28","arxiv_id":"2505.22170","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-adversarial-analysis-of","title":"A Framework for Adversarial Analysis of Decision Support Systems Prior to Deployment","date":"2025-05-27","arxiv_id":"2505.21414","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-treatment-strategies-for-1","title":"Learning optimal treatment strategies for intraoperative hypotension using deep reinforcement learning","date":"2025-05-27","arxiv_id":"2505.21596","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-control-improves-residential","title":"Algorithmic Control Improves Residential Building Energy and EV Management when PV Capacity is High but Battery Capacity is Low","date":"2025-05-26","arxiv_id":"2505.20377","repositories_listed":0,"syntology":null},{"url":null,"slug":"reduce-computational-cost-in-deep","title":"Reduce Computational Cost In Deep Reinforcement Learning Via Randomized Policy Learning","date":"2025-05-25","arxiv_id":"2505.19054","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-modular-framework-for-automated-evaluation","title":"A modular framework for automated evaluation of procedural content generation in serious games with deep reinforcement learning agents","date":"2025-05-22","arxiv_id":"2505.16801","repositories_listed":0,"syntology":null},{"url":null,"slug":"backdoors-in-drl-four-environments-focusing","title":"Backdoors in DRL: Four Environments Focusing on In-distribution Triggers","date":"2025-05-22","arxiv_id":"2505.17248","repositories_listed":0,"syntology":null},{"url":null,"slug":"find-the-fruit-designing-a-zero-shot-sim2real","title":"Find the Fruit: Designing a Zero-Shot Sim2Real Deep RL Planner for Occlusion Aware Plant Manipulation","date":"2025-05-22","arxiv_id":"2505.16547","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-deep-reinforcement-learning","title":"Energy-Efficient Deep Reinforcement Learning with Spiking Transformers","date":"2025-05-20","arxiv_id":"2505.14533","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-via-focused-satisficing","title":"Imitation Learning via Focused Satisficing","date":"2025-05-20","arxiv_id":"2505.14820","repositories_listed":0,"syntology":null}],"record_sha256":"f9fe3ed6487f1b5612eb2a237ce443c82683a046eab89ce53962c42e57d5469e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}