{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/30","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":30,"pages_in_order":59,"rows_per_page":100,"rows":[2901,3000],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/29","next":"/task/deep-reinforcement-learning/papers/31","papers":[{"url":null,"slug":"robust-computation-offloading-and-trajectory","title":"Robust Computation Offloading and Trajectory Optimization for Multi-UAV-Assisted MEC: A Multi-Agent DRL Approach","date":"2023-08-24","arxiv_id":"2308.12756","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-14","title":"A Mobile Data-Driven Hierarchical Deep Reinforcement Learning Approach for Real-time Demand-Responsive Railway Rescheduling and Station Overcrowding Mitigation","date":"2023-08-23","arxiv_id":"2308.11849","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobility-aware-computation-offloading-for","title":"Mobility-Aware Computation Offloading for Swarm Robotics using Deep Reinforcement Learning","date":"2023-08-22","arxiv_id":"2308.11154","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-deep-reinforcement-learning-approach","title":"A Safe Deep Reinforcement Learning Approach for Energy Efficient Federated Learning in Wireless Communication Networks","date":"2023-08-21","arxiv_id":"2308.10664","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-artificial","title":"Deep Reinforcement Learning for Artificial Upwelling Energy Management","date":"2023-08-20","arxiv_id":"2308.10199","repositories_listed":0,"syntology":null},{"url":null,"slug":"ilcas-imitation-learning-based-configuration","title":"ILCAS: Imitation Learning-Based Configuration-Adaptive Streaming for Live Video Analytics with Cross-Camera Collaboration","date":"2023-08-19","arxiv_id":"2308.10068","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-computational-efficient-bots-with","title":"Learning Computational Efficient Bots with Costly Features","date":"2023-08-18","arxiv_id":"2308.09629","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-language-models-for-drl","title":"Leveraging Large Language Models for DRL-Based Anti-Jamming Strategies in Zero Touch Networks","date":"2023-08-18","arxiv_id":"2308.09376","repositories_listed":0,"syntology":null},{"url":null,"slug":"reduced-order-modeling-of-a-moose-based","title":"Reduced Order Modeling of a MOOSE-based Advanced Manufacturing Model with Operator Learning","date":"2023-08-18","arxiv_id":"2308.09691","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-for-electric","title":"Federated Reinforcement Learning for Electric Vehicles Charging Control on Distribution Networks","date":"2023-08-17","arxiv_id":"2308.08792","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-robust-integrated-multi-strategy-bus","title":"A Robust Integrated Multi-Strategy Bus Control System via Deep Reinforcement Learning","date":"2023-08-16","arxiv_id":"2308.08179","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-renewable-energy-in-agriculture-a","title":"Integrating Renewable Energy in Agriculture: A Deep Reinforcement Learning-based Approach","date":"2023-08-16","arxiv_id":"2308.08611","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-process-2","title":"Deep reinforcement learning for process design: Review and perspective","date":"2023-08-15","arxiv_id":"2308.07822","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-tracking-of-a-single-rigid-body","title":"Adaptive Tracking of a Single-Rigid-Body Character in Various Environments","date":"2023-08-14","arxiv_id":"2308.07491","repositories_listed":0,"syntology":null},{"url":null,"slug":"covernav-cover-following-navigation-planning","title":"CoverNav: Cover Following Navigation Planning in Unstructured Outdoor Environment with Deep Reinforcement Learning","date":"2023-08-12","arxiv_id":"2308.06594","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-team-based-navigation-a-review-of","title":"Learning Team-Based Navigation: A Review of Deep Reinforcement Learning Techniques for Multi-Agent Pathfinding","date":"2023-08-11","arxiv_id":"2308.05893","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-classical-and-deep","title":"A Comparison of Classical and Deep Reinforcement Learning Methods for HVAC Control","date":"2023-08-10","arxiv_id":"2308.05711","repositories_listed":0,"syntology":null},{"url":null,"slug":"commodities-trading-through-deep-policy","title":"Commodities Trading through Deep Policy Gradient Methods","date":"2023-08-10","arxiv_id":"2309.00630","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-deep-reinforcement-learning-for-1","title":"Adversarial Deep Reinforcement Learning for Cyber Security in Software Defined Networks","date":"2023-08-09","arxiv_id":"2308.04909","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-type-multi-agent-deep","title":"Cooperative Multi-Type Multi-Agent Deep Reinforcement Learning for Resource Management in Space-Air-Ground Integrated Networks","date":"2023-08-08","arxiv_id":"2308.03995","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-diverse-data-types","title":"Deep Learning for Steganalysis of Diverse Data Types: A review of methods, taxonomy, challenges and future directions","date":"2023-08-08","arxiv_id":"2308.04522","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-360-degree-videos-in-metaverse","title":"Heterogeneous 360 Degree Videos in Metaverse: Differentiated Reinforcement Learning Approaches","date":"2023-08-08","arxiv_id":"2308.04083","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-drl-method-for-fast-solution-of-real","title":"A Safe DRL Method for Fast Solution of Real-Time Optimal Power Flow","date":"2023-08-07","arxiv_id":"2308.03420","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligence-endogenous-management-platform","title":"Intelligence-Endogenous Management Platform for Computing and Network Convergence","date":"2023-08-07","arxiv_id":"2308.03450","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-the-switching-operation-in","title":"Optimizing the switching operation in monoclonal antibody production: Economic MPC and reinforcement learning","date":"2023-08-07","arxiv_id":"2308.03928","repositories_listed":0,"syntology":null},{"url":null,"slug":"surrogate-empowered-sim2real-transfer-of-deep","title":"Surrogate Empowered Sim2Real Transfer of Deep Reinforcement Learning for ORC Superheat Control","date":"2023-08-05","arxiv_id":"2308.02765","repositories_listed":0,"syntology":null},{"url":null,"slug":"vehicles-control-collision-avoidance-using","title":"Vehicles Control: Collision Avoidance using Federated Deep Reinforcement Learning","date":"2023-08-04","arxiv_id":"2308.02614","repositories_listed":0,"syntology":null},{"url":null,"slug":"avoidance-navigation-based-on-offline-pre","title":"Avoidance Navigation Based on Offline Pre-Training Reinforcement Learning","date":"2023-08-03","arxiv_id":"2308.01551","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-the-solo12-quadruped-robot-with","title":"Controlling the Solo12 Quadruped Robot with Deep Reinforcement Learning","date":"2023-08-02","arxiv_id":"2309.16683","repositories_listed":0,"syntology":null},{"url":null,"slug":"caching-at-stars-the-next-generation-edge","title":"Caching-at-STARS: the Next Generation Edge Caching","date":"2023-08-01","arxiv_id":"2308.00562","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-battery","title":"Deep Reinforcement Learning-Based Battery Conditioning Hierarchical V2G Coordination for Multi-Stakeholder Benefits","date":"2023-08-01","arxiv_id":"2308.00218","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-goal-audio-visual-navigation-using","title":"Multi-goal Audio-visual Navigation using Sound Direction Map","date":"2023-08-01","arxiv_id":"2308.00219","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-search-and-navigation-in-heterogeneous","title":"Target Search and Navigation in Heterogeneous Robot Systems with Deep Reinforcement Learning","date":"2023-08-01","arxiv_id":"2308.00331","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-safety-filter-for","title":"Distributionally Robust Safety Filter for Learning-Based Control in Active Distribution Systems","date":"2023-07-31","arxiv_id":"2307.16351","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-deep-reinforcement-learning-algorithm","title":"Dynamic deep-reinforcement-learning algorithm in Partially Observed Markov Decision Processes","date":"2023-07-29","arxiv_id":"2307.15931","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-payload-thermal-control","title":"Autonomous Payload Thermal Control","date":"2023-07-28","arxiv_id":"2307.15438","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ensemble-method-of-deep-reinforcement","title":"An Ensemble Method of Deep Reinforcement Learning for Automated Cryptocurrency Trading","date":"2023-07-27","arxiv_id":"2309.00626","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-safety-constraints-in","title":"Evaluation of Safety Constraints in Autonomous Navigation with Deep Reinforcement Learning","date":"2023-07-27","arxiv_id":"2307.14568","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robust-goal","title":"Deep Reinforcement Learning for Robust Goal-Based Wealth Management","date":"2023-07-25","arxiv_id":"2307.13501","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-uav-speed-control-with-collision","title":"Multi-UAV Speed Control with Collision Avoidance and Handover-aware Cell Association: DRL with Action Branching","date":"2023-07-24","arxiv_id":"2307.13158","repositories_listed":0,"syntology":null},{"url":null,"slug":"qamplifynet-pushing-the-boundaries-of-supply","title":"QAmplifyNet: Pushing the Boundaries of Supply Chain Backorder Prediction Using Interpretable Hybrid Quantum-Classical Neural Network","date":"2023-07-24","arxiv_id":"2307.12906","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-control-of-flow-over-rotating-cylinder","title":"Active Control of Flow over Rotating Cylinder by Multiple Jets using Deep Reinforcement Learning","date":"2023-07-22","arxiv_id":"2307.12083","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-system-for","title":"Deep Reinforcement Learning Based System for Intraoperative Hyperspectral Video Autofocusing","date":"2023-07-21","arxiv_id":"2307.11638","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-control-of-resource-flow-to-optimize","title":"Adaptive Control of Resource Flow to Optimize Construction Work and Cash Flow via Online Deep Reinforcement Learning","date":"2023-07-20","arxiv_id":"2307.10574","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-decision-making-framework-for-recommended","title":"A Decision Making Framework for Recommended Maintenance of Road Segments","date":"2023-07-19","arxiv_id":"2307.10085","repositories_listed":0,"syntology":null},{"url":null,"slug":"a3d-adaptive-accurate-and-autonomous","title":"A3D: Adaptive, Accurate, and Autonomous Navigation for Edge-Assisted Drones","date":"2023-07-19","arxiv_id":"2307.09880","repositories_listed":0,"syntology":null},{"url":null,"slug":"amortised-design-optimization-for-item","title":"Amortised Design Optimization for Item Response Theory","date":"2023-07-19","arxiv_id":"2307.09891","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-service-caching-communication-and","title":"Joint Service Caching, Communication and Computing Resource Allocation in Collaborative MEC Systems: A DRL-based Two-timescale Approach","date":"2023-07-19","arxiv_id":"2307.09691","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-driving-policy-learning-with-guided","title":"Robust Driving Policy Learning with Guided Meta Reinforcement Learning","date":"2023-07-19","arxiv_id":"2307.10160","repositories_listed":0,"syntology":null},{"url":null,"slug":"mining-of-single-class-by-active-learning-for","title":"Mining of Single-Class by Active Learning for Semantic Segmentation","date":"2023-07-18","arxiv_id":"2307.09109","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-multiagent-flexibility-aggregation","title":"A Novel Multiagent Flexibility Aggregation Framework","date":"2023-07-17","arxiv_id":"2307.08401","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-the-dynamic","title":"Deep reinforcement learning for the dynamic vehicle dispatching problem: An event-based approach","date":"2023-07-13","arxiv_id":"2307.07508","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-control-policy-for-artificial-pancreas","title":"Hybrid Control Policy for Artificial Pancreas via Ensemble Deep Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.06501","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-from-distributed-machine-learning-to","title":"A Survey From Distributed Machine Learning to Distributed Deep Learning","date":"2023-07-11","arxiv_id":"2307.05232","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiobjective-hydropower-reservoir-operation","title":"Multiobjective Hydropower Reservoir Operation Optimization with Transformer-Based Deep Reinforcement Learning","date":"2023-07-11","arxiv_id":"2307.05643","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-trajectory-optimization-for-directional","title":"UAV Trajectory Optimization for Directional THz Links Using Deep Reinforcement Learning","date":"2023-07-08","arxiv_id":"2307.05535","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-and-deep-reinforcement-learning","title":"Reinforcement and Deep Reinforcement Learning-based Solutions for Machine Maintenance Planning, Scheduling Policies, and Optimization","date":"2023-07-07","arxiv_id":"2307.03860","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-the-robustness-of-qmix-against","title":"Enhancing the Robustness of QMIX against State-adversarial Attacks","date":"2023-07-03","arxiv_id":"2307.00907","repositories_listed":0,"syntology":null},{"url":null,"slug":"ga-drl-graph-neural-network-augmented-deep","title":"GA-DRL: Graph Neural Network-Augmented Deep Reinforcement Learning for DAG Task Scheduling over Dynamic Vehicular Clouds","date":"2023-07-03","arxiv_id":"2307.00777","repositories_listed":0,"syntology":null},{"url":null,"slug":"asset-correlation-based-deep-reinforcement","title":"Asset correlation based deep reinforcement learning for the portfolio selection","date":"2023-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-multi-agent-deep-reinforcement-1","title":"Federated Multi-Agent Deep Reinforcement Learning for Dynamic and Flexible 3D Operation of 5G Multi-MAP Networks","date":"2023-06-30","arxiv_id":"2307.06842","repositories_listed":0,"syntology":null},{"url":"/paper/resetting-the-optimizer-in-deep-rl-an","slug":"resetting-the-optimizer-in-deep-rl-an","title":"Resetting the Optimizer in Deep RL: An Empirical Study","date":"2023-06-30","arxiv_id":"2306.17833","repositories_listed":0,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/resetting-the-optimizer-in-deep-rl-an#ran","syntology_url":"https://syntology.ai/paper/2306.17833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17833"}},"official":null}},{"url":null,"slug":"eigensubspace-of-temporal-difference-dynamics","title":"Eigensubspace of Temporal-Difference Dynamics and How It Improves Value Approximation in Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.16750","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-important-sensory-feedback-for","title":"Identifying Important Sensory Feedback for Learning Locomotion Skills","date":"2023-06-29","arxiv_id":"2306.17101","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-environment-models-with-continuous","title":"Learning Environment Models with Continuous Stochastic Dynamics","date":"2023-06-29","arxiv_id":"2306.17204","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-is-strength-mastering-football-full","title":"Diversity is Strength: Mastering Football Full Game with Interactive Reinforcement Learning of Multiple AIs","date":"2023-06-28","arxiv_id":"2306.15903","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-in-reinforcement-learning-a-survey","title":"Structure in Deep Reinforcement Learning: A Survey and Open Problems","date":"2023-06-28","arxiv_id":"2306.16021","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-effective-task-offloading-scheduling-for","title":"Cost-Effective Task Offloading Scheduling for Hybrid Mobile Edge-Quantum Computing","date":"2023-06-26","arxiv_id":"2306.14588","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-12","title":"Multi-Agent Deep Reinforcement Learning for Dynamic Avatar Migration in AIoT-enabled Vehicular Metaverses with Trajectory Prediction","date":"2023-06-26","arxiv_id":"2306.14683","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-critical-scenario-generation-via","title":"Safety-Critical Scenario Generation Via Reinforcement Learning Based Editing","date":"2023-06-25","arxiv_id":"2306.14131","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-semantics-aware-communication","title":"Task-Oriented Semantics-Aware Communication for Wireless UAV Control and Command Transmission","date":"2023-06-25","arxiv_id":"2306.14228","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-q-transformer-visual-explanation-in","title":"Action Q-Transformer: Visual Explanation in Deep Reinforcement Learning with Encoder-Decoder Model using Action Query","date":"2023-06-24","arxiv_id":"2306.13879","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-13","title":"Multi-agent Deep Reinforcement Learning for Distributed Load Restoration","date":"2023-06-24","arxiv_id":"2306.14018","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-multi-agent-reinforcement-5","title":"Decentralized Multi-Agent Reinforcement Learning with Global State Prediction","date":"2023-06-22","arxiv_id":"2306.12926","repositories_listed":0,"syntology":null},{"url":null,"slug":"mp3-movement-primitive-based-re-planning","title":"MP3: Movement Primitive-Based (Re-)Planning Policy","date":"2023-06-22","arxiv_id":"2306.12729","repositories_listed":0,"syntology":null},{"url":null,"slug":"sum-rate-maximization-of-rsma-based-aerial","title":"Sum-Rate Maximization of RSMA-based Aerial Communications with Energy Harvesting: A Reinforcement Learning Approach","date":"2023-06-22","arxiv_id":"2306.12977","repositories_listed":0,"syntology":null},{"url":null,"slug":"introspective-action-advising-for","title":"Introspective Action Advising for Interpretable Transfer Learning","date":"2023-06-21","arxiv_id":"2306.12314","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-driving-with-deep-reinforcement","title":"Autonomous Driving with Deep Reinforcement Learning in CARLA Simulation","date":"2023-06-20","arxiv_id":"2306.11217","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-strategy-guided-reinforcement","title":"Evolutionary Strategy Guided Reinforcement Learning via MultiBuffer Communication","date":"2023-06-20","arxiv_id":"2306.11535","repositories_listed":0,"syntology":null},{"url":null,"slug":"int-hrl-towards-intention-based-hierarchical","title":"Int-HRL: Towards Intention-based Hierarchical Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11483","repositories_listed":0,"syntology":null},{"url":null,"slug":"inter-cell-network-slicing-with-transfer","title":"Inter-Cell Network Slicing With Transfer Learning Empowered Multi-Agent Deep Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11552","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-efficient-comfort-and-energy-saving","title":"Safe, Efficient, Comfort, and Energy-saving Automated Driving through Roundabout Based on Deep Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11465","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-robustness-of-deep-reinforcement","title":"Benchmarking Robustness of Deep Reinforcement Learning approaches to Online Portfolio Management","date":"2023-06-19","arxiv_id":"2306.10950","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-esg-financial","title":"Deep Reinforcement Learning for ESG financial portfolio management","date":"2023-06-19","arxiv_id":"2307.09631","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-tick-level-data-and-periodical","title":"Integrating Tick-level Data and Periodical Signal for High-frequency Market Making","date":"2023-06-19","arxiv_id":"2306.17179","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmpr-enhancing-opportunistic-routing-in","title":"DeepMPR: Enhancing Opportunistic Routing in Wireless Networks through Multi-Agent Deep Reinforcement Learning","date":"2023-06-16","arxiv_id":"2306.09637","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-open-ran-slice-management","title":"Attention-based Open RAN Slice Management using Deep Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09490","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-curriculum-training-for-drl","title":"Evolutionary Curriculum Training for DRL-Based Navigation Systems","date":"2023-06-15","arxiv_id":"2306.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"inroads-into-autonomous-network-defence-using","title":"Inroads into Autonomous Network Defence using Explained Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-maneuver-planning-with-deep","title":"Predictive Maneuver Planning with Deep Reinforcement Learning (PMP-DRL) for comfortable and safe autonomous driving","date":"2023-06-15","arxiv_id":"2306.09055","repositories_listed":0,"syntology":null},{"url":null,"slug":"carbon-emissions-and-sustainability-of","title":"Carbon emissions and sustainability of launching 5G mobile networks in China","date":"2023-06-14","arxiv_id":"2306.08337","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-policy-gradient-methods-in-commodity","title":"Deep Policy Gradient Methods in Commodity Markets","date":"2023-06-14","arxiv_id":"2308.01910","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-interval-restrictions-on-action","title":"Dynamic Interval Restrictions on Action Spaces in Deep Reinforcement Learning for Obstacle Avoidance","date":"2023-06-13","arxiv_id":"2306.08008","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-market-energy-optimization-with","title":"Multi-market Energy Optimization with Renewables via Reinforcement Learning","date":"2023-06-13","arxiv_id":"2306.08147","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-collision-momentum-in-deep","title":"Using Collision Momentum in Deep Reinforcement Learning Based Adversarial Pedestrian Modeling","date":"2023-06-13","arxiv_id":"2306.07525","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeptransition-viability-leads-to-the","title":"DeepTransition: Viability Leads to the Emergence of Gait Transitions in Learning Anticipatory Quadrupedal Locomotion Skills","date":"2023-06-12","arxiv_id":"2306.07419","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-testing-scenario-generation-method","title":"Evolving Testing Scenario Generation Method and Intelligence Evaluation Framework for Automated Vehicles","date":"2023-06-12","arxiv_id":"2306.07142","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-guided-by","title":"Multi-Agent Reinforcement Learning Guided by Signal Temporal Logic Specifications","date":"2023-06-11","arxiv_id":"2306.06808","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-trajectory-and-multi-user-beamforming","title":"UAV Trajectory and Multi-User Beamforming Optimization for Clustered Users Against Passive Eavesdropping Attacks With Unknown CSI","date":"2023-06-11","arxiv_id":"2306.06686","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-adversarial-directions-in-deep","title":"Detecting Adversarial Directions in Deep Reinforcement Learning to Make Robust Decisions","date":"2023-06-09","arxiv_id":"2306.05873","repositories_listed":0,"syntology":null}],"record_sha256":"503b1e9060f3a987d7005ec41d6c902381f8317d2f971e82c8c1d2aa523198c0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}