{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/20","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":59,"rows_per_page":100,"rows":[1901,2000],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/19","next":"/task/deep-reinforcement-learning/papers/21","papers":[{"url":null,"slug":"deep-reinforcement-learning-based-video","title":"Deep Reinforcement Learning-based Video-Haptic Radio Resource Slicing in Tactile Internet","date":"2025-03-18","arxiv_id":"2503.14066","repositories_listed":0,"syntology":null},{"url":null,"slug":"safeslice-enabling-sla-compliant-o-ran","title":"SafeSlice: Enabling SLA-Compliant O-RAN Slicing via Safe Deep Reinforcement Learning","date":"2025-03-17","arxiv_id":"2503.12753","repositories_listed":0,"syntology":null},{"url":null,"slug":"emobipednav-emotion-aware-social-navigation","title":"EmoBipedNav: Emotion-aware Social Navigation for Bipedal Robots with Deep Reinforcement Learning","date":"2025-03-16","arxiv_id":"2503.12538","repositories_listed":0,"syntology":null},{"url":null,"slug":"movable-cell-free-massive-mimo-for-high-speed","title":"Movable Cell-Free Massive MIMO For High-Speed Train Communications: A PPO-Based Antenna Position Optimization","date":"2025-03-16","arxiv_id":"2503.12405","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-torque-control-of-exoskeletons-under","title":"Adaptive Torque Control of Exoskeletons under Spasticity Conditions via Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11433","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-competitive-and-collusive-behaviors","title":"Exploring Competitive and Collusive Behaviors in Algorithmic Pricing with Deep Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11270","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-cost-real-world-implementation-of-the","title":"Low-cost Real-world Implementation of the Swing-up Pendulum for Deep Reinforcement Learning Experiments","date":"2025-03-14","arxiv_id":"2503.11065","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-directional-locomotion-for","title":"Training Directional Locomotion for Quadrupedal Low-Cost Robotic Systems via Deep Reinforcement Learning","date":"2025-03-14","arxiv_id":"2503.11059","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-nonlinear-real-time-capable-motion-cueing","title":"A nonlinear real time capable motion cueing algorithm based on deep reinforcement learning","date":"2025-03-13","arxiv_id":"2503.10419","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-to-10","title":"A Deep Reinforcement Learning Approach to Automated Stock Trading, using xLSTM Networks","date":"2025-03-12","arxiv_id":"2503.09655","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-guided-reinforcement-learning-policy","title":"Rule-Guided Reinforcement Learning Policy Evaluation and Improvement","date":"2025-03-12","arxiv_id":"2503.09270","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-locomotion-transformer-with","title":"Unified Locomotion Transformer with Simultaneous Sim-to-Real Transfer for Quadrupeds","date":"2025-03-12","arxiv_id":"2503.08997","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-soc-in-battery-cells-using-safe","title":"Balancing SoC in Battery Cells using Safe Action Perturbations","date":"2025-03-11","arxiv_id":"2503.11696","repositories_listed":0,"syntology":null},{"url":null,"slug":"beam-selection-in-isac-using-contextual","title":"Beam Selection in ISAC using Contextual Bandit with Multi-modal Transformer and Transfer Learning","date":"2025-03-11","arxiv_id":"2503.08937","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-driven-control-of-bioelectric-signalling","title":"AI-driven control of bioelectric signalling for real-time topological reorganization of cells","date":"2025-03-10","arxiv_id":"2503.13489","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-effective-design-of-grid-tied-community","title":"Cost-Effective Design of Grid-tied Community Microgrid","date":"2025-03-10","arxiv_id":"2503.07414","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-and-design-on-intelligent","title":"Research and Design on Intelligent Recognition of Unordered Targets for Robots Based on Reinforcement Learning","date":"2025-03-10","arxiv_id":"2503.07340","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-spectrum-sharing-in-integrated-tn","title":"Intelligent Spectrum Sharing in Integrated TN-NTNs: A Hierarchical Deep Reinforcement Learning Approach","date":"2025-03-09","arxiv_id":"2503.06720","repositories_listed":0,"syntology":null},{"url":null,"slug":"pull-based-query-scheduling-for-goal-oriented","title":"Pull-Based Query Scheduling for Goal-Oriented Semantic Communication","date":"2025-03-09","arxiv_id":"2503.06725","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultho-ultra-lightweight-yet-efficient","title":"ULTHO: Ultra-Lightweight yet Efficient Hyperparameter Optimization in Deep Reinforcement Learning","date":"2025-03-08","arxiv_id":"2503.06101","repositories_listed":0,"syntology":null},{"url":null,"slug":"guaranteeing-out-of-distribution-detection-in","title":"Guaranteeing Out-Of-Distribution Detection in Deep RL via Transition Estimation","date":"2025-03-07","arxiv_id":"2503.05238","repositories_listed":0,"syntology":null},{"url":null,"slug":"impoola-the-power-of-average-pooling-for","title":"Impoola: The Power of Average Pooling for Image-Based Deep Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05546","repositories_listed":0,"syntology":null},{"url":null,"slug":"aolo-analysis-and-optimization-for-low-carbon","title":"AOLO: Analysis and Optimization For Low-Carbon Oriented Wireless Large Language Model Services","date":"2025-03-06","arxiv_id":"2503.04418","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-optimize-deep-rl-policy-weights-as","title":"Can We Optimize Deep RL Policy Weights as Trajectory Modeling?","date":"2025-03-06","arxiv_id":"2503.04074","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-inverse-q-learning-from","title":"Multi-Agent Inverse Q-Learning from Demonstrations","date":"2025-03-06","arxiv_id":"2503.04679","repositories_listed":0,"syntology":null},{"url":null,"slug":"palo-learning-posture-aware-locomotion-for","title":"PALo: Learning Posture-Aware Locomotion for Quadruped Robots","date":"2025-03-06","arxiv_id":"2503.04462","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-is-more-rewards-in-rl-for-cyber-defence","title":"Less is more? Rewards in RL for Cyber Defence","date":"2025-03-05","arxiv_id":"2503.03245","repositories_listed":0,"syntology":null},{"url":null,"slug":"koopman-based-generalization-of-deep","title":"Koopman-Based Generalization of Deep Reinforcement Learning With Application to Wireless Communications","date":"2025-03-04","arxiv_id":"2503.02961","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01069","title":"Multi-Agent Reinforcement Learning with Long-Term Performance Objectives for Service Workforce Optimization","date":"2025-03-03","arxiv_id":"2503.01069","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-04803","title":"An energy-efficient learning solution for the Agile Earth Observation Satellite Scheduling Problem","date":"2025-03-03","arxiv_id":"2503.04803","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-user-1","title":"Deep Reinforcement Learning-Based User Association in Hybrid LiFi/WiFi Indoor Networks","date":"2025-03-03","arxiv_id":"2503.01803","repositories_listed":0,"syntology":null},{"url":null,"slug":"stone-soup-multi-target-tracking-feature","title":"Stone Soup Multi-Target Tracking Feature Extraction For Autonomous Search And Track In Deep Reinforcement Learning Environment","date":"2025-03-03","arxiv_id":"2503.01293","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-deep-learning-in-financial-asset","title":"The Role of Deep Learning in Financial Asset Management: A Systematic Review","date":"2025-03-03","arxiv_id":"2503.01591","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-00331","title":"PINN-DT: Optimizing Energy Consumption in Smart Building Using Hybrid Physics-Informed Neural Networks and Digital Twin Framework with Blockchain Security","date":"2025-03-01","arxiv_id":"2503.00331","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-laser-pulses-with-reinforcement","title":"Shaping Laser Pulses with Reinforcement Learning","date":"2025-03-01","arxiv_id":"2503.00499","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-priority-aware-5g-nr-u-wi-fi","title":"Traffic Priority-Aware 5G NR-U/Wi-Fi Coexistence with Deep Reinforcement Learning","date":"2025-03-01","arxiv_id":"2503.00256","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-and-modular-network-on-non","title":"Hierarchical and Modular Network on Non-prehensile Manipulation in General Environments","date":"2025-02-28","arxiv_id":"2502.20843","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deterministic-policy-gradient-for","title":"Robust Deterministic Policy Gradient for Disturbance Attenuation and Its Application to Quadrotor Control","date":"2025-02-28","arxiv_id":"2502.21057","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generative-model-enhanced-multi-agent","title":"A Generative Model Enhanced Multi-Agent Reinforcement Learning Method for Electric Vehicle Charging Navigation","date":"2025-02-27","arxiv_id":"2502.20068","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-autonomous","title":"Deep Reinforcement Learning based Autonomous Decision-Making for Cooperative UAVs: A Search and Rescue Real World Application","date":"2025-02-27","arxiv_id":"2502.20326","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-efficiency-of-a-deep","title":"Improving the Efficiency of a Deep Reinforcement Learning-Based Power Management System for HPC Clusters Using Curriculum Learning","date":"2025-02-27","arxiv_id":"2502.20348","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-drl-based-framework-for-optimal","title":"A Multi-Agent DRL-Based Framework for Optimal Resource Allocation and Twin Migration in the Multi-Tier Vehicular Metaverse","date":"2025-02-26","arxiv_id":"2502.19004","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-secure-spectrum-reuse-d2d","title":"DRL-Based Secure Spectrum-Reuse D2D Communications with RIS Assistance","date":"2025-02-26","arxiv_id":"2502.18742","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-auto-encoders-for-enhanced-deep","title":"Recurrent Auto-Encoders for Enhanced Deep Reinforcement Learning in Wilderness Search and Rescue Planning","date":"2025-02-26","arxiv_id":"2502.19356","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-on-edge-computing-and-cloud","title":"Research on Edge Computing and Cloud Collaborative Resource Scheduling Optimization Based on Deep Reinforcement Learning","date":"2025-02-26","arxiv_id":"2502.18773","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-deep-reinforcement-learning-1","title":"Applications of deep reinforcement learning to urban transit network design","date":"2025-02-25","arxiv_id":"2502.17758","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-performance-bounds-for-digital-twin","title":"Provable Performance Bounds for Digital Twin-driven Deep Reinforcement Learning in Wireless Networks: A Novel Digital-Twin Bisimulation Metric","date":"2025-02-25","arxiv_id":"2502.17983","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-defense-with-multiple-defenders-and-an","title":"Target Defense with Multiple Defenders and an Agile Attacker via Residual Policy Learning","date":"2025-02-25","arxiv_id":"2502.18549","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-6-dof-autonomous-underwater-vehicle","title":"Toward 6-DOF Autonomous Underwater Vehicle Energy-Aware Position Control based on Deep Reinforcement Learning: Preliminary Results","date":"2025-02-25","arxiv_id":"2502.17742","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-coordination-for-heterogeneous","title":"Distributed Coordination for Heterogeneous Non-Terrestrial Networks","date":"2025-02-24","arxiv_id":"2502.17366","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-based-limit-order-book-simulation-under","title":"Event-Based Limit Order Book Simulation under a Neural Hawkes Process: Application in Market-Making","date":"2025-02-24","arxiv_id":"2502.17417","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-liquidity-aware-bond-yields-using","title":"Predicting Liquidity-Aware Bond Yields using Causal GANs and Deep Reinforcement Learning with LLM Evaluation","date":"2025-02-24","arxiv_id":"2502.17011","repositories_listed":0,"syntology":null},{"url":null,"slug":"facilitating-emergency-vehicle-passage-in","title":"Facilitating Emergency Vehicle Passage in Congested Urban Areas Using Multi-agent Deep Reinforcement Learning","date":"2025-02-23","arxiv_id":"2502.16449","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-sentiment-manipulation-by-llm","title":"Exploring Sentiment Manipulation by LLM-Enabled Intelligent Trading Agents","date":"2025-02-22","arxiv_id":"2502.16343","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-ai-collaboration-in-cloud-security","title":"Human-AI Collaboration in Cloud Security: Cognitive Hierarchy-Driven Deep Reinforcement Learning","date":"2025-02-22","arxiv_id":"2502.16054","repositories_listed":0,"syntology":null},{"url":"/paper/hyperspherical-normalization-for-scalable","slug":"hyperspherical-normalization-for-scalable","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15280","repositories_listed":0,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyperspherical-normalization-for-scalable#ran","syntology_url":"https://syntology.ai/paper/2502.15280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15280"}},"official":null}},{"url":null,"slug":"salsa-rl-stability-analysis-in-the-latent","title":"SALSA-RL: Stability Analysis in the Latent Space of Actions for Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.15512","repositories_listed":0,"syntology":null},{"url":null,"slug":"spikerl-a-scalable-and-energy-efficient","title":"SpikeRL: A Scalable and Energy-efficient Framework for Deep Spiking Reinforcement Learning","date":"2025-02-21","arxiv_id":"2502.17496","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-ultrasound-image","title":"Reinforcement Learning for Ultrasound Image Analysis A Comprehensive Review of Advances and Applications","date":"2025-02-20","arxiv_id":"2502.14995","repositories_listed":0,"syntology":null},{"url":null,"slug":"sprig-stackelberg-perception-reinforcement","title":"SPRIG: Stackelberg Perception-Reinforcement Learning with Internal Game Dynamics","date":"2025-02-20","arxiv_id":"2502.14264","repositories_listed":0,"syntology":null},{"url":null,"slug":"atomic-proximal-policy-optimization-for","title":"Atomic Proximal Policy Optimization for Electric Robo-Taxi Dispatch and Charger Allocation","date":"2025-02-19","arxiv_id":"2502.13392","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-collision-free-success-rate-for","title":"Improving Collision-Free Success Rate For Object Goal Visual Navigation Via Two-Stage Training With Collision Prediction","date":"2025-02-19","arxiv_id":"2502.13498","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-target-radar-search-and-track-using","title":"Multi-Target Radar Search and Track Using Sequence-Capable Deep Reinforcement Learning","date":"2025-02-19","arxiv_id":"2502.13584","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-enhanced-deep-reinforcement-learning","title":"A Graph-Enhanced Deep-Reinforcement Learning Framework for the Aircraft Landing Problem","date":"2025-02-18","arxiv_id":"2502.12617","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-sim-to-real-methods-in-rl","title":"A Survey of Sim-to-Real Methods in RL: Progress, Prospects and Challenges with Foundation Models","date":"2025-02-18","arxiv_id":"2502.13187","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-strategy-on-macro-and-micro","title":"Communication Strategy on Macro-and-Micro Traffic State in Cooperative Deep Reinforcement Learning for Regional Traffic Signal Control","date":"2025-02-18","arxiv_id":"2502.13248","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-optimal-trading-history-in","title":"Finding Optimal Trading History in Reinforcement Learning for Stock Market Trading","date":"2025-02-18","arxiv_id":"2502.12537","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-high-quality-robotic-wiping-policy","title":"Learning a High-quality Robotic Wiping Policy Using Systematic Reward Analysis and Visual-Language Model Based Curriculum","date":"2025-02-18","arxiv_id":"2502.12599","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntp-int-network-traffic-prediction-driven-in","title":"NTP-INT: Network Traffic Prediction-Driven In-band Network Telemetry for High-load Switches","date":"2025-02-18","arxiv_id":"2502.12834","repositories_listed":0,"syntology":null},{"url":null,"slug":"hovering-flight-of-soft-actuated-insect-scale","title":"Hovering Flight of Soft-Actuated Insect-Scale Micro Aerial Vehicles using Deep Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.12355","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-mobile-ai-generated-content","title":"Intelligent Mobile AI-Generated Content Services via Interactive Prompt Engineering and Dynamic Service Provisioning","date":"2025-02-17","arxiv_id":"2502.11386","repositories_listed":0,"syntology":null},{"url":null,"slug":"massively-scaling-explicit-policy-conditioned","title":"Massively Scaling Explicit Policy-conditioned Value Functions","date":"2025-02-17","arxiv_id":"2502.11949","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-deformable-object-manipulation-via-nmpc","title":"Robot Deformable Object Manipulation via NMPC-generated Demonstrations in Deep Reinforcement Learning","date":"2025-02-17","arxiv_id":"2502.11375","repositories_listed":0,"syntology":null},{"url":null,"slug":"tss-gaz-ptp-towards-improving-gumbel","title":"TSS GAZ PTP: Towards Improving Gumbel AlphaZero with Two-stage Self-play for Multi-constrained Electric Vehicle Routing Problems","date":"2025-02-17","arxiv_id":"2502.15777","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-bidding","title":"Deep Reinforcement Learning-Based Bidding Strategies for Prosumers Trading in Double Auction-Based Transactive Energy Market","date":"2025-02-16","arxiv_id":"2502.15774","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-language-models-for-enhanced","title":"Integrating Language Models for Enhanced Network State Monitoring in DRL-Based SFC Provisioning","date":"2025-02-16","arxiv_id":"2502.11298","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-online-resource-constrained","title":"Solving Online Resource-Constrained Scheduling for Follow-Up Observation in Astronomy: a Reinforcement Learning Approach","date":"2025-02-16","arxiv_id":"2502.11134","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-bottleneck-reinforcement-learning-joint","title":"Rule-Bottleneck Reinforcement Learning: Joint Explanation and Decision Optimization for Resource Allocation with Language Agents","date":"2025-02-15","arxiv_id":"2502.10732","repositories_listed":0,"syntology":null},{"url":null,"slug":"aoi-sensitive-data-forwarding-with","title":"AoI-Sensitive Data Forwarding with Distributed Beamforming in UAV-Assisted IoT","date":"2025-02-13","arxiv_id":"2502.09038","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-user","title":"Deep Reinforcement Learning-Based User Scheduling for Collaborative Perception","date":"2025-02-12","arxiv_id":"2502.10456","repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-the-training-efficiency-of-drl-for","title":"Improve the Training Efficiency of DRL for Wireless Communication Resource Allocation: The Role of Generative Diffusion Models","date":"2025-02-11","arxiv_id":"2502.07211","repositories_listed":0,"syntology":null},{"url":null,"slug":"migt-memory-instance-gated-transformer","title":"MIGT: Memory Instance Gated Transformer Framework for Financial Portfolio Management","date":"2025-02-11","arxiv_id":"2502.07280","repositories_listed":0,"syntology":null},{"url":null,"slug":"picts-a-novel-deep-reinforcement-learning","title":"PICTS: A Novel Deep Reinforcement Learning Approach for Dynamic P-I Control in Scanning Probe Microscopy","date":"2025-02-11","arxiv_id":"2502.07326","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-joint-mobile-edge-computing-and","title":"UAV-assisted Joint Mobile Edge Computing and Data Collection via Matching-enabled Deep Reinforcement Learning","date":"2025-02-11","arxiv_id":"2502.07388","repositories_listed":0,"syntology":null},{"url":null,"slug":"satisfaction-aware-incentive-scheme-for","title":"Meta-Computing Enhanced Federated Learning in IIoT: Satisfaction-Aware Incentive Scheme via DRL-Based Stackelberg Game","date":"2025-02-10","arxiv_id":"2502.06909","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-offloading-in-vehicular-edge-computing","title":"Intelligent Offloading in Vehicular Edge Computing: A Comprehensive Review of Deep Reinforcement Learning Approaches and Architectures","date":"2025-02-10","arxiv_id":"2502.06963","repositories_listed":0,"syntology":null},{"url":null,"slug":"aerial-reliable-collaborative-communications","title":"Aerial Reliable Collaborative Communications for Terrestrial Mobile Users via Evolutionary Multi-Objective Deep Reinforcement Learning","date":"2025-02-09","arxiv_id":"2502.05824","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-control-in-multi-rotor-aerial-robots","title":"Motion Control in Multi-Rotor Aerial Robots Using Deep Reinforcement Learning","date":"2025-02-09","arxiv_id":"2502.05996","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-explainable-deep-reinforcement","title":"A Survey on Explainable Deep Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.06869","repositories_listed":0,"syntology":null},{"url":null,"slug":"closing-the-responsibility-gap-in-ai-based","title":"Closing the Responsibility Gap in AI-based Network Management: An Intelligent Audit System Approach","date":"2025-02-08","arxiv_id":"2502.05608","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-paradigm-in-tuning-learned-indexes-a","title":"A New Paradigm in Tuning Learned Indexes: A Reinforcement Learning Enhanced Approach","date":"2025-02-07","arxiv_id":"2502.05001","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-adaptive-anti-jamming-channel-access-via","title":"Fast Adaptive Anti-Jamming Channel Access via Deep Q Learning and Coarse-Grained Spectrum Prediction","date":"2025-02-07","arxiv_id":"2502.04963","repositories_listed":0,"syntology":null},{"url":null,"slug":"seasonal-station-keeping-of-short-duration","title":"Seasonal Station-Keeping of Short Duration High Altitude Balloons using Deep Reinforcement Learning","date":"2025-02-07","arxiv_id":"2502.05014","repositories_listed":0,"syntology":null},{"url":null,"slug":"stride-automating-reward-design-deep","title":"STRIDE: Automating Reward Design, Deep Reinforcement Learning Training and Feedback Optimization in Humanoid Robotics Locomotion","date":"2025-02-07","arxiv_id":"2502.04692","repositories_listed":0,"syntology":null},{"url":null,"slug":"illuminating-spaces-deep-reinforcement","title":"Illuminating Spaces: Deep Reinforcement Learning and Laser-Wall Partitioning for Architectural Layout Generation","date":"2025-02-06","arxiv_id":"2502.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-3","title":"Deep Reinforcement Learning-Based Optimization of Second-Life Battery Utilization in Electric Vehicles Charging Stations","date":"2025-02-05","arxiv_id":"2502.03412","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-authenticated-task-offloading-in","title":"Lightweight Authenticated Task Offloading in 6G-Cloud Vehicular Twin Networks","date":"2025-02-05","arxiv_id":"2502.03403","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-hiding-and-smart-anti-jamming","title":"Achieving Hiding and Smart Anti-Jamming Communication: A Parallel DRL Approach against Moving Reactive Jammer","date":"2025-02-04","arxiv_id":"2502.02385","repositories_listed":0,"syntology":null},{"url":null,"slug":"magnnet-multi-agent-graph-neural-network","title":"MAGNNET: Multi-Agent Graph Neural Network-based Efficient Task Allocation for Autonomous Vehicles with Deep Reinforcement Learning","date":"2025-02-04","arxiv_id":"2502.02311","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-optimized-portfolio-enhancement","title":"Regret-Optimized Portfolio Enhancement through Deep Reinforcement Learning and Future Looking Rewards","date":"2025-02-04","arxiv_id":"2502.02619","repositories_listed":0,"syntology":null}],"record_sha256":"9e0bf73a6ef7a25751012e2724d209711b9418d164a534bcb64def6c1631ce21","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}