{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/25","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":59,"rows_per_page":100,"rows":[2401,2500],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/24","next":"/task/deep-reinforcement-learning/papers/26","papers":[{"url":"/paper/optimizing-automatic-differentiation-with","slug":"optimizing-automatic-differentiation-with","title":"Optimizing Automatic Differentiation with Deep Reinforcement Learning","date":"2024-06-07","arxiv_id":"2406.05027","repositories_listed":0,"syntology":{"n":29,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":17,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 17 unverified","sample_list":"/paper/optimizing-automatic-differentiation-with#ran","syntology_url":"https://syntology.ai/paper/2406.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05027"}},"official":null}},{"url":null,"slug":"sim-to-real-transfer-of-deep-reinforcement","title":"Sim-to-Real Transfer of Deep Reinforcement Learning Agents for Online Coverage Path Planning","date":"2024-06-07","arxiv_id":"2406.04920","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-pessimism-and-optimism-dynamics-in","title":"Exploring Pessimism and Optimism Dynamics in Deep Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.03890","repositories_listed":0,"syntology":null},{"url":null,"slug":"gensafe-a-generalizable-safety-enhancer-for","title":"GenSafe: A Generalizable Safety Enhancer for Safe Reinforcement Learning Algorithms Based on Reduced Order Markov Decision Process Model","date":"2024-06-06","arxiv_id":"2406.03912","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-dynamic-network-utility","title":"Stochastic Dynamic Network Utility Maximization with Application to Disaster Response","date":"2024-06-06","arxiv_id":"2406.03750","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generalized-apprenticeship-learning","title":"A Generalized Apprenticeship Learning Framework for Modeling Heterogeneous Student Pedagogical Strategies","date":"2024-06-04","arxiv_id":"2406.02450","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-collusion-in-dynamic-pricing-with","title":"Algorithmic Collusion in Dynamic Pricing with Deep Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02437","repositories_listed":0,"syntology":null},{"url":null,"slug":"by-fair-means-or-foul-quantifying-collusion","title":"By Fair Means or Foul: Quantifying Collusion in a Market Simulation with Deep Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02650","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-in-aerial-and","title":"Improving Generalization in Aerial and Terrestrial Mobile Robots Control Through Delayed Policy Learning","date":"2024-06-04","arxiv_id":"2406.01952","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifying-the-generalization-of-deep-learning","title":"Verifying the Generalization of Deep Learning to Out-of-Distribution Domains","date":"2024-06-04","arxiv_id":"2406.02024","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-drl-agents-in-commercial-fighting","title":"Advancing DRL Agents in Commercial Fighting Games: Training, Integration, and Agent-Human Alignment","date":"2024-06-03","arxiv_id":"2406.01103","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-advanced-reinforcement-learning-framework","title":"An Advanced Reinforcement Learning Framework for Online Scheduling of Deferrable Workloads in Cloud Computing","date":"2024-06-03","arxiv_id":"2406.01047","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-behavioral-mode","title":"Deep Reinforcement Learning Behavioral Mode Switching Using Optimal Control Based on a Latent Space Objective","date":"2024-06-03","arxiv_id":"2406.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-weakly","title":"Deep reinforcement learning for weakly coupled MDP's with continuous actions","date":"2024-06-03","arxiv_id":"2406.01099","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-assisted-optimal-bidding","title":"Large Language Model Assisted Optimal Bidding of BESS in FCAS Market: An AI-agent based Approach","date":"2024-06-03","arxiv_id":"2406.00974","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-meets-leaf","title":"Multi-Agent Reinforcement Learning Meets Leaf Sequencing in Radiotherapy","date":"2024-06-03","arxiv_id":"2406.01853","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-digital-twin-framework-for-reinforcement","title":"A Digital Twin Framework for Reinforcement Learning with Real-Time Self-Improvement via Human Assistive Teleoperation","date":"2024-06-02","arxiv_id":"2406.00732","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-on-the-application-of-computer","title":"Research on the Application of Computer Vision Based on Deep Learning in Autonomous Driving Technology","date":"2024-06-01","arxiv_id":"2406.00490","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-learning-foundation-models-for","title":"Towards Learning Foundation Models for Heuristic Functions to Solve Pathfinding Problems","date":"2024-06-01","arxiv_id":"2406.02598","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ai-for-deep-reinforcement-learning","title":"Generative AI for Deep Reinforcement Learning: Framework, Analysis, and Use Cases","date":"2024-05-31","arxiv_id":"2405.20568","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-oriented-sensor-reporting-scheduling-for","title":"Goal-Oriented Sensor Reporting Scheduling for Non-linear Dynamic System Monitoring","date":"2024-05-31","arxiv_id":"2405.20983","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-20","title":"A Deep Reinforcement Learning Approach for Trading Optimization in the Forex Market with Multi-Agent Asynchronous Distribution","date":"2024-05-30","arxiv_id":"2405.19982","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-battlefield-awareness-an-aerial-ris","title":"Enhancing Battlefield Awareness: An Aerial RIS-assisted ISAC System with Deep Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.20168","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-as-a-monotone-scheme","title":"Q-learning as a monotone scheme","date":"2024-05-30","arxiv_id":"2405.20538","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-household-robotics-deep-interactive","title":"Advancing Household Robotics: Deep Interactive Reinforcement Learning for Efficient Training and Enhanced Performance","date":"2024-05-29","arxiv_id":"2405.18687","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-impact-of-traffic-signal","title":"Exploring the impact of traffic signal control and connected and automated vehicles on intersections safety: A deep reinforcement learning approach","date":"2024-05-29","arxiv_id":"2405.19236","repositories_listed":0,"syntology":null},{"url":null,"slug":"proactive-load-shaping-strategies-with","title":"Proactive Load-Shaping Strategies with Privacy-Cost Trade-offs in Residential Households based on Deep Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.18888","repositories_listed":0,"syntology":null},{"url":null,"slug":"mollification-effects-of-policy-gradient","title":"Mollification Effects of Policy Gradient Methods","date":"2024-05-28","arxiv_id":"2405.17832","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-models-for-general-surgical-grasping","title":"World Models for General Surgical Grasping","date":"2024-05-28","arxiv_id":"2405.17940","repositories_listed":0,"syntology":null},{"url":null,"slug":"biological-neurons-compete-with-deep","title":"Biological Neurons Compete with Deep Reinforcement Learning in Sample Efficiency in a Simulated Gameworld","date":"2024-05-27","arxiv_id":"2405.16946","repositories_listed":0,"syntology":null},{"url":null,"slug":"amortized-active-causal-induction-with-deep","title":"Amortized Active Causal Induction with Deep Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16718","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-q-network-on-the-fly-target","slug":"adaptive-q-network-on-the-fly-target","title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16195","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-q-network-on-the-fly-target#ran","syntology_url":"https://syntology.ai/paper/2405.16195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16195"}},"official":null}},{"url":null,"slug":"counterexample-guided-repair-of-reinforcement","title":"Counterexample-Guided Repair of Reinforcement Learning Systems Using Safety Critics","date":"2024-05-24","arxiv_id":"2405.15430","repositories_listed":0,"syntology":null},{"url":null,"slug":"sf-dqn-provable-knowledge-transfer-using","title":"SF-DQN: Provable Knowledge Transfer using Successor Feature for Deep Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15920","repositories_listed":0,"syntology":null},{"url":null,"slug":"transmission-interface-power-flow-adjustment","title":"Transmission Interface Power Flow Adjustment: A Deep Reinforcement Learning Approach based on Multi-task Attribution Map","date":"2024-05-24","arxiv_id":"2405.15831","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-behavior-aware-approach-for-deep","title":"A Behavior-Aware Approach for Deep Reinforcement Learning in Non-stationary Environments without Known Change Points","date":"2024-05-23","arxiv_id":"2405.14214","repositories_listed":0,"syntology":null},{"url":null,"slug":"closed-form-symbolic-solutions-a-new","title":"Closed-form Symbolic Solutions: A New Perspective on Solving Partial Differential Equations","date":"2024-05-23","arxiv_id":"2405.14620","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-5-5","title":"Deep Reinforcement Learning for 5*5 Multiplayer Go","date":"2024-05-23","arxiv_id":"2405.14265","repositories_listed":0,"syntology":null},{"url":null,"slug":"doubly-dynamic-isac-precoding-for-vehicular","title":"Doubly-Dynamic ISAC Precoding for Vehicular Networks: A Constrained Deep Reinforcement Learning (CDRL) Approach","date":"2024-05-23","arxiv_id":"2405.14347","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-time-critical","title":"Deep Reinforcement Learning for Time-Critical Wilderness Search And Rescue Using Drones","date":"2024-05-21","arxiv_id":"2405.12800","repositories_listed":0,"syntology":null},{"url":null,"slug":"gase-graph-attention-sampling-with-edges","title":"GASE: Graph Attention Sampling with Edges Fusion for Solving Vehicle Routing Problems","date":"2024-05-21","arxiv_id":"2405.12475","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-field-spot-beamfocusing-a-correlation","title":"Near-Field Spot Beamfocusing: A Correlation-Aware Transfer Learning Approach","date":"2024-05-21","arxiv_id":"2405.19347","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-robustness-assessment-adversarial","title":"Rethinking Robustness Assessment: Adversarial Attacks on Learning-based Quadrupedal Locomotion Controllers","date":"2024-05-21","arxiv_id":"2405.12424","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-deep-reinforcement-learning-for","title":"Continual Deep Reinforcement Learning for Decentralized Satellite Routing","date":"2024-05-20","arxiv_id":"2405.12308","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-impact-of-choice-on-deep","title":"Investigating the Impact of Choice on Deep Reinforcement Learning for Space Controls","date":"2024-05-20","arxiv_id":"2405.12355","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-dive-into-model-free-reinforcement","title":"Deep Dive into Model-free Reinforcement Learning for Biological and Robotic Systems: Theory and Practice","date":"2024-05-19","arxiv_id":"2405.11457","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-vehicle-aerodynamics-with-deep","title":"Enhancing Vehicle Aerodynamics with Deep Reinforcement Learning in Voxelised Models","date":"2024-05-19","arxiv_id":"2405.11492","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-distributional-value-functions-for","title":"Exploiting Distributional Value Functions for Financial Market Valuation, Enhanced Feature Creation and Improvement of Trading Algorithms","date":"2024-05-19","arxiv_id":"2405.11686","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-learning-with-energy-harvesting","title":"Federated Learning With Energy Harvesting Devices: An MDP Framework","date":"2024-05-17","arxiv_id":"2405.10513","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-transfer-learning-for-uav","title":"Continuous Transfer Learning for UAV Communication-aware Trajectory Design","date":"2024-05-16","arxiv_id":"2405.10087","repositories_listed":0,"syntology":null},{"url":null,"slug":"chaos-based-reinforcement-learning-with-td3","title":"Chaos-based reinforcement learning with TD3","date":"2024-05-15","arxiv_id":"2405.09086","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-continuous-integration-skip-a","title":"Detecting Continuous Integration Skip : A Reinforcement Learning-based Approach","date":"2024-05-15","arxiv_id":"2405.09657","repositories_listed":0,"syntology":null},{"url":null,"slug":"dvs-rg-differential-variable-speed-limits","title":"DVS-RG: Differential Variable Speed Limits Control using Deep Reinforcement Learning with Graph State Representation","date":"2024-05-15","arxiv_id":"2405.09163","repositories_listed":0,"syntology":null},{"url":null,"slug":"cier-a-novel-experience-replay-approach-with","title":"CIER: A Novel Experience Replay Approach with Causal Inference in Deep Reinforcement Learning","date":"2024-05-14","arxiv_id":"2405.08380","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-real-time-2","title":"Deep Reinforcement Learning for Real-Time Ground Delay Program Revision and Corresponding Flight Delay Assignments","date":"2024-05-14","arxiv_id":"2405.08298","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-deep-reinforcement-learning-for","title":"Optimizing Deep Reinforcement Learning for American Put Option Hedging","date":"2024-05-14","arxiv_id":"2405.08602","repositories_listed":0,"syntology":null},{"url":null,"slug":"madrl-based-rate-adaptation-for-360-degree","title":"MADRL-Based Rate Adaptation for 360° Video Streaming with Multi-Viewpoint Prediction","date":"2024-05-13","arxiv_id":"2405.07759","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-demand-model-and-client-deployment-in","title":"On-Demand Model and Client Deployment in Federated Learning with Deep Reinforcement Learning","date":"2024-05-12","arxiv_id":"2405.07175","repositories_listed":0,"syntology":null},{"url":null,"slug":"auditing-an-automatic-grading-model-with-deep","title":"Auditing an Automatic Grading Model with deep Reinforcement Learning","date":"2024-05-11","arxiv_id":"2405.07087","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealthy-imitation-reward-guided-environment","title":"Stealthy Imitation: Reward-guided Environment-free Policy Stealing","date":"2024-05-11","arxiv_id":"2405.07004","repositories_listed":0,"syntology":null},{"url":null,"slug":"hedging-american-put-options-with-deep","title":"Hedging American Put Options with Deep Reinforcement Learning","date":"2024-05-10","arxiv_id":"2405.06774","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-multi-user-rf","title":"Deep Reinforcement Learning for Multi-User RF Charging with Non-linear Energy Harvesters","date":"2024-05-07","arxiv_id":"2405.04218","repositories_listed":0,"syntology":null},{"url":null,"slug":"latency-and-energy-minimization-in-noma","title":"Latency and Energy Minimization in NOMA-Assisted MEC Network: A Federated Deep Reinforcement Learning Approach","date":"2024-05-07","arxiv_id":"2405.04012","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-reinforcement-learning-of-curative","title":"End-to-End Reinforcement Learning of Curative Curtailment with Partial Measurement Availability","date":"2024-05-06","arxiv_id":"2405.03262","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-o-ran-security-evasion-attacks-and","title":"Enhancing O-RAN Security: Evasion Attacks and Robust Defenses for Graph Reinforcement Learning-based Connection Management","date":"2024-05-06","arxiv_id":"2405.03891","repositories_listed":0,"syntology":null},{"url":null,"slug":"guidance-design-for-escape-flight-vehicle","title":"Guidance Design for Escape Flight Vehicle Using Evolution Strategy Enhanced Deep Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.03711","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-safe-set-algorithm-for-provably-safe","title":"Implicit Safe Set Algorithm for Provably Safe Reinforcement Learning","date":"2024-05-04","arxiv_id":"2405.02754","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robot-soccer-from-egocentric-vision","title":"Learning Robot Soccer from Egocentric Vision with Deep Reinforcement Learning","date":"2024-05-03","arxiv_id":"2405.02425","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-imitation-for-manipulator-control","title":"Behavior Imitation for Manipulator Control and Grasping with Deep Reinforcement Learning","date":"2024-05-02","arxiv_id":"2405.01284","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-active-learning-for-the-search-of","title":"Generative Active Learning for the Search of Small-molecule Protein Binders","date":"2024-05-02","arxiv_id":"2405.01616","repositories_listed":0,"syntology":null},{"url":null,"slug":"tabular-and-deep-reinforcement-learning-for","title":"Tabular and Deep Reinforcement Learning for Gittins Index","date":"2024-05-02","arxiv_id":"2405.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"hugo-highlighting-unseen-grid-options","title":"HUGO -- Highlighting Unseen Grid Options: Combining Deep Reinforcement Learning with a Heuristic Target Topology Approach","date":"2024-05-01","arxiv_id":"2405.00629","repositories_listed":0,"syntology":null},{"url":null,"slug":"portfolio-management-using-deep-reinforcement","title":"Portfolio Management using Deep Reinforcement Learning","date":"2024-05-01","arxiv_id":"2405.01604","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-advanced","title":"Deep Reinforcement Learning for Advanced Longitudinal Control and Collision Avoidance in High-Risk Driving Scenarios","date":"2024-04-29","arxiv_id":"2404.19087","repositories_listed":0,"syntology":null},{"url":null,"slug":"shared-learning-of-powertrain-control","title":"Shared learning of powertrain control policies for vehicle fleets","date":"2024-04-27","arxiv_id":"2404.17892","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-explainable-deep-reinforcement-learning","title":"An Explainable Deep Reinforcement Learning Model for Warfarin Maintenance Dosing Using Policy Distillation and Action Forging","date":"2024-04-26","arxiv_id":"2404.17187","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl2fc-an-attack-resilient-controller-for","title":"DRL2FC: An Attack-Resilient Controller for Automatic Generation Control Based on Deep Reinforcement Learning","date":"2024-04-25","arxiv_id":"2404.16974","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-dynamics-of-data-transmission","title":"Exploring the Dynamics of Data Transmission in 5G Networks: A Conceptual Analysis","date":"2024-04-25","arxiv_id":"2404.16508","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-high-speed-cruising-performance-of","title":"Enhancing High-Speed Cruising Performance of Autonomous Vehicles through Integrated Deep Reinforcement Learning Framework","date":"2024-04-23","arxiv_id":"2404.14713","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-deep-reinforcement-learning-2","title":"Multi-Objective Deep Reinforcement Learning for 5G Base Station Placement to Support Localisation for Future Sustainable Traffic","date":"2024-04-23","arxiv_id":"2404.14954","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-to-promote","title":"Using deep reinforcement learning to promote sustainable human behaviour on a common pool resource problem","date":"2024-04-23","arxiv_id":"2404.15059","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-coordination-of-distributed","title":"Decentralized Coordination of Distributed Energy Resources through Local Energy Markets and Deep Reinforcement Learning","date":"2024-04-19","arxiv_id":"2404.13142","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-aided","title":"Deep Reinforcement Learning-aided Transmission Design for Energy-efficient Link Optimization in Vehicular Communications","date":"2024-04-19","arxiv_id":"2404.12595","repositories_listed":0,"syntology":null},{"url":null,"slug":"random-network-distillation-based-deep","title":"Random Network Distillation Based Deep Reinforcement Learning for AGV Path Planning","date":"2024-04-19","arxiv_id":"2404.12594","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-policy-optimization-with-temporal-logic","title":"LTL-Constrained Policy Optimization with Cycle Experience Replay","date":"2024-04-17","arxiv_id":"2404.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"eyeformer-predicting-personalized-scanpaths","title":"EyeFormer: Predicting Personalized Scanpaths with Transformer-Guided Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.10163","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-intelligent-optimization-algorithms","title":"Advanced Intelligent Optimization Algorithms for Multi-Objective Optimal Power Flow in Future Power Systems: A Review","date":"2024-04-14","arxiv_id":"2404.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"codecloak-a-method-for-evaluating-and","title":"CodeCloak: A Method for Evaluating and Mitigating Code Leakage by LLM Code Assistants","date":"2024-04-13","arxiv_id":"2404.09066","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-online","title":"Deep Reinforcement Learning based Online Scheduling Policy for Deep Neural Network Multi-Tenant Multi-Accelerator Systems","date":"2024-04-13","arxiv_id":"2404.08950","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-forest-fire-prevention-deep","title":"Advancing Forest Fire Prevention: Deep Reinforcement Learning for Effective Firebreak Placement","date":"2024-04-12","arxiv_id":"2404.08523","repositories_listed":0,"syntology":null},{"url":null,"slug":"anti-byzantine-attacks-enabled-vehicle","title":"Anti-Byzantine Attacks Enabled Vehicle Selection for Asynchronous Federated Learning in Vehicular Edge Computing","date":"2024-04-12","arxiv_id":"2404.08444","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-configuring-exploration-exploitation","title":"Auto-configuring Exploration-Exploitation Tradeoff in Evolutionary Computation via Deep Reinforcement Learning","date":"2024-04-12","arxiv_id":"2404.08239","repositories_listed":0,"syntology":null},{"url":null,"slug":"kinematics-modeling-of-peroxy-free-radicals-a","title":"Kinematics Modeling of Peroxy Free Radicals: A Deep Reinforcement Learning Approach","date":"2024-04-12","arxiv_id":"2404.10010","repositories_listed":0,"syntology":null},{"url":null,"slug":"prescribing-optimal-health-aware-operation","title":"Prescribing Optimal Health-Aware Operation for Urban Air Mobility with Deep Reinforcement Learning","date":"2024-04-12","arxiv_id":"2404.08497","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlemmo-evolutionary-multimodal-optimization","title":"RLEMMO: Evolutionary Multimodal Optimization Assisted By Deep Reinforcement Learning","date":"2024-04-12","arxiv_id":"2404.08242","repositories_listed":0,"syntology":null},{"url":null,"slug":"tdanet-target-directed-attention-network-for","title":"TDANet: Target-Directed Attention Network For Object-Goal Visual Navigation With Zero-Shot Ability","date":"2024-04-12","arxiv_id":"2404.08353","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-ground-space-communications-via","title":"Collaborative Ground-Space Communications via Evolutionary Multi-objective Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.07450","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-divide-and-conquer-placement-using-deep","title":"FPGA Divide-and-Conquer Placement using Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.13061","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-probabilistic-planning-for","title":"Generative Probabilistic Planning for Optimizing Supply Chain Networks","date":"2024-04-11","arxiv_id":"2404.07511","repositories_listed":0,"syntology":null},{"url":null,"slug":"r2-indicator-and-deep-reinforcement-learning","title":"R2 Indicator and Deep Reinforcement Learning Enhanced Adaptive Multi-Objective Evolutionary Algorithm","date":"2024-04-11","arxiv_id":"2404.08161","repositories_listed":0,"syntology":null}],"record_sha256":"e216fd00e42aebd90ad7b63e257a12892095d79e755ea8d3d777b1d7476a4bfb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}