{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/24","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":24,"pages_in_order":59,"rows_per_page":100,"rows":[2301,2400],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/23","next":"/task/deep-reinforcement-learning/papers/25","papers":[{"url":null,"slug":"2408-03084","title":"Research on Autonomous Driving Decision-making Strategies based Deep Reinforcement Learning","date":"2024-08-06","arxiv_id":"2408.03084","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-aware-consistent-edge-selection","title":"Communication-Aware Consistent Edge Selection for Mobile Users and Autonomous Vehicles","date":"2024-08-06","arxiv_id":"2408.03435","repositories_listed":0,"syntology":null},{"url":null,"slug":"screener-a-general-framework-for-task","title":"SCREENER: A general framework for task-specific experiment design in quantitative MRI","date":"2024-08-06","arxiv_id":"2408.11834","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01979","title":"Shaping Rewards, Shaping Routes: On Multi-Agent Deep Q-Networks for Routing in Satellite Constellation Networks","date":"2024-08-04","arxiv_id":"2408.01979","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02022","title":"Scenario-based Thermal Management Parametrization Through Deep Reinforcement Learning","date":"2024-08-04","arxiv_id":"2408.02022","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02047","title":"Latency-Aware Resource Allocation for Mobile Edge Generation and Computing via Deep Reinforcement Learning","date":"2024-08-04","arxiv_id":"2408.02047","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01656","title":"Deep Reinforcement Learning for Dynamic Order Picking in Warehouse Operations","date":"2024-08-03","arxiv_id":"2408.01656","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01188","title":"Multi-Objective Deep Reinforcement Learning for Optimisation in Autonomous Systems","date":"2024-08-02","arxiv_id":"2408.01188","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00814","title":"Adaptive traffic signal safety and efficiency improvement by multi objective deep reinforcement learning approach","date":"2024-08-01","arxiv_id":"2408.00814","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-llm-enabled-in-context","title":"Large Language Model (LLM)-enabled In-context Learning for Wireless Network Optimization: A Case Study of Power Control","date":"2024-08-01","arxiv_id":"2408.00214","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-transit-signal-priority-based-on","title":"Adaptive Transit Signal Priority based on Deep Reinforcement Learning and Connected Vehicles in a Traffic Microsimulation Environment","date":"2024-07-31","arxiv_id":"2408.00098","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-based-deep-reinforcement-learning-with","title":"Image-Based Deep Reinforcement Learning with Intrinsically Motivated Stimuli: On the Execution of Complex Robotic Tasks","date":"2024-07-31","arxiv_id":"2407.21338","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-machine-learning-architecture-search","title":"Quantum Machine Learning Architecture Search via Deep Reinforcement Learning","date":"2024-07-29","arxiv_id":"2407.20147","repositories_listed":0,"syntology":null},{"url":null,"slug":"sustainable-task-offloading-in-secure-uav","title":"Sustainable Task Offloading in Secure UAV-assisted Smart Farm Networks: A Multi-Agent DRL with Action Mask Approach","date":"2024-07-29","arxiv_id":"2407.19657","repositories_listed":0,"syntology":null},{"url":null,"slug":"reputation-driven-asynchronous-federated","title":"Reputation-Driven Asynchronous Federated Learning for Enhanced Trajectory Prediction with Blockchain","date":"2024-07-28","arxiv_id":"2407.19428","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-interpretability-of-codebooks-in-model","title":"The Interpretability of Codebooks in Model-Based Reinforcement Learning is Limited","date":"2024-07-28","arxiv_id":"2407.19532","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-human-like","title":"Large Language Models for Human-like Autonomous Driving: A Survey","date":"2024-07-27","arxiv_id":"2407.19280","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-18","title":"Multi-Agent Deep Reinforcement Learning for Energy Efficient Multi-Hop STAR-RIS-Assisted Transmissions","date":"2024-07-26","arxiv_id":"2407.18627","repositories_listed":0,"syntology":null},{"url":null,"slug":"shangus-deep-reinforcement-learning-meets","title":"FH-DRL: Exponential-Hyperbolic Frontier Heuristics with DRL for accelerated Exploration in Unknown Environments","date":"2024-07-26","arxiv_id":"2407.18892","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-17","title":"Multi-Agent Deep Reinforcement Learning for Resilience Optimization in 5G RAN","date":"2024-07-25","arxiv_id":"2407.18066","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-and-context-aware-route-planning","title":"Personalized and Context-aware Route Planning for Edge-assisted Vehicles","date":"2024-07-25","arxiv_id":"2407.17980","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-drl-for-urllc-in-open-ran-a-trade","title":"Harnessing DRL for URLLC in Open RAN: A Trade-off Exploration","date":"2024-07-24","arxiv_id":"2407.17598","repositories_listed":0,"syntology":null},{"url":null,"slug":"movelight-enhancing-traffic-signal-control","title":"MoveLight: Enhancing Traffic Signal Control through Movement-Centric Deep Reinforcement Learning","date":"2024-07-24","arxiv_id":"2407.17303","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-based-bandit-a-medium-access","title":"Neural Network-Based Bandit: A Medium Access Control for the IIoT Alarm Scenario","date":"2024-07-23","arxiv_id":"2407.16877","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-model-based-resource-allocation","title":"Diffusion Model Based Resource Allocation Strategy in Ultra-Reliable Wireless Networked Control Systems","date":"2024-07-22","arxiv_id":"2407.15784","repositories_listed":0,"syntology":null},{"url":null,"slug":"modrl-ta-a-multi-objective-deep-reinforcement","title":"MODRL-TA:A Multi-Objective Deep Reinforcement Learning Framework for Traffic Allocation in E-Commerce Search","date":"2024-07-22","arxiv_id":"2407.15476","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-imitation-learning-through-graph","title":"Offline Imitation Learning Through Graph Search and Retrieval","date":"2024-07-22","arxiv_id":"2407.15403","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-deep-reinforcement-learning","title":"Mitigating Deep Reinforcement Learning Backdoors in the Neural Activation Space","date":"2024-07-21","arxiv_id":"2407.15168","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-and-bridging-the-gap-between","title":"Analyzing and Bridging the Gap between Maximizing Total Reward and Discounted Reward in Deep Reinforcement Learning","date":"2024-07-18","arxiv_id":"2407.13279","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-multi-2","title":"Deep Reinforcement Learning for Multi-Objective Optimization: Enhancing Wind Turbine Energy Generation while Mitigating Noise Emissions","date":"2024-07-18","arxiv_id":"2407.13320","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepclair-utilizing-market-forecasts-for","title":"DeepClair: Utilizing Market Forecasts for Effective Portfolio Selection","date":"2024-07-18","arxiv_id":"2407.13427","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-triggered-reinforcement-learning-based","title":"Event-Triggered Reinforcement Learning Based Joint Resource Allocation for Ultra-Reliable Low-Latency V2X Communications","date":"2024-07-18","arxiv_id":"2407.13947","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiobjective-vehicle-routing-optimization","title":"Multiobjective Vehicle Routing Optimization with Time Windows: A Hybrid Approach Using Deep Reinforcement Learning and NSGA-II","date":"2024-07-18","arxiv_id":"2407.13113","repositories_listed":0,"syntology":null},{"url":null,"slug":"random-latent-exploration-for-deep","title":"Random Latent Exploration for Deep Reinforcement Learning","date":"2024-07-18","arxiv_id":"2407.13755","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-tutorial-and-survey","title":"Reinforcement Learning: Tutorial and Survey","date":"2024-07-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-reaction-barriers-with-deep","title":"Estimating Reaction Barriers with Deep Reinforcement Learning","date":"2024-07-17","arxiv_id":"2407.12453","repositories_listed":0,"syntology":null},{"url":null,"slug":"maintenance-strategies-for-sewer-pipes-with","title":"Maintenance Strategies for Sewer Pipes with Multi-State Degradation and Deep Reinforcement Learning","date":"2024-07-17","arxiv_id":"2407.12894","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-joint-resource-scheduling-of-embb","title":"DRL-based Joint Resource Scheduling of eMBB and URLLC in O-RAN","date":"2024-07-16","arxiv_id":"2407.11558","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynsyn-dynamical-synergistic-representation","title":"DynSyn: Dynamical Synergistic Representation for Efficient Learning and Control in Overactuated Embodied Systems","date":"2024-07-16","arxiv_id":"2407.11472","repositories_listed":0,"syntology":null},{"url":null,"slug":"green-resource-allocation-in-cloud-native-o","title":"Green Resource Allocation in Cloud-Native O-RAN Enabled Small Cell Networks","date":"2024-07-16","arxiv_id":"2407.11563","repositories_listed":0,"syntology":null},{"url":null,"slug":"satisficing-exploration-for-deep","title":"Satisficing Exploration for Deep Reinforcement Learning","date":"2024-07-16","arxiv_id":"2407.12185","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-operators-for","title":"Deep Learning-Based Operators for Evolutionary Algorithms","date":"2024-07-15","arxiv_id":"2407.10477","repositories_listed":0,"syntology":null},{"url":null,"slug":"soc-boundary-and-battery-aging-aware","title":"SOC-Boundary and Battery Aging Aware Hierarchical Coordination of Multiple EV Aggregates Among Multi-stakeholders with Multi-Agent Constrained Deep Reinforcement Learning","date":"2024-07-14","arxiv_id":"2407.13790","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-symmetric-1","title":"Deep reinforcement learning with symmetric data augmentation applied for aircraft lateral attitude tracking control","date":"2024-07-13","arxiv_id":"2407.11077","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-distortion-canceling-and-control","title":"Model-free Distortion Canceling and Control of Quantum Devices","date":"2024-07-13","arxiv_id":"2407.09877","repositories_listed":0,"syntology":null},{"url":null,"slug":"pail-performance-based-adversarial-imitation","title":"PAIL: Performance based Adversarial Imitation Learning Engine for Carbon Neutral Optimization","date":"2024-07-12","arxiv_id":"2407.08910","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-based","title":"Distributed Deep Reinforcement Learning Based Gradient Quantization for Federated Learning Enabled Vehicle Edge Computing","date":"2024-07-11","arxiv_id":"2407.08462","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancements-in-recommender-systems-a","title":"Advancements in Recommender Systems: A Comprehensive Analysis Based on Data, Algorithms, and Evaluation","date":"2024-07-10","arxiv_id":"2407.18937","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-llms-to-explain-drl-decisions-for","title":"Leveraging LLMs to explain DRL decisions for transparent 6G network slicing","date":"2024-07-10","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-fair-star-ris-for-mobile","title":"Energy Efficient Fair STAR-RIS for Mobile Users","date":"2024-07-09","arxiv_id":"2407.06868","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-open-source-multi-agent-deep-reinforcement","title":"An open source Multi-Agent Deep Reinforcement Learning Routing Simulator for satellite networks","date":"2024-07-08","arxiv_id":"2407.11047","repositories_listed":0,"syntology":null},{"url":null,"slug":"wastewater-treatment-plant-data-for-nutrient","title":"Wastewater Treatment Plant Data for Nutrient Removal System","date":"2024-07-07","arxiv_id":"2407.05346","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-bifurcation-method-for-observation","title":"A Novel Bifurcation Method for Observation Perturbation Attacks on Reinforcement Learning Agents: Load Altering Attacks on a Cyber Physical Power System","date":"2024-07-06","arxiv_id":"2407.05182","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-generative-diffusion-models-for-attack","title":"Hybrid-Generative Diffusion Models for Attack-Oriented Twin Migration in Vehicular Metaverses","date":"2024-07-05","arxiv_id":"2407.11036","repositories_listed":0,"syntology":null},{"url":null,"slug":"question-answering-with-texts-and-tables","title":"Question Answering with Texts and Tables through Deep Reinforcement Learning","date":"2024-07-05","arxiv_id":"2407.04858","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-quantization-and-pruning-on","title":"The Impact of Quantization and Pruning on Deep Reinforcement Learning Models","date":"2024-07-05","arxiv_id":"2407.04803","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-time-scale-service-caching-and-pricing","title":"Multi-Time Scale Service Caching and Pricing in MEC Systems with Dynamic Program Popularity","date":"2024-07-04","arxiv_id":"2407.03804","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-role-of-environmental-complexity-on","title":"A Role of Environmental Complexity on Representation Learning in Deep Reinforcement Learning Agents","date":"2024-07-03","arxiv_id":"2407.03436","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-decision-making-for-multi-user-360","title":"Multi-Task Decision-Making for Multi-User 360 Video Processing over Wireless Networks","date":"2024-07-03","arxiv_id":"2407.03426","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-autopilot-constrained-drl-for","title":"Adaptive Autopilot: Constrained DRL for Diverse Driving Behaviors","date":"2024-07-02","arxiv_id":"2407.02546","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-decoupling-capacitor","title":"Hierarchical Decoupling Capacitor Optimization for Power Distribution Network of 2.5D ICs with Co-Analysis of Frequency and Time Domains Based on Deep Reinforcement Learning","date":"2024-07-02","arxiv_id":"2407.04737","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-to-8","title":"A Deep Reinforcement Learning Approach to Battery Management in Dairy Farming via Proximal Policy Optimization","date":"2024-07-01","arxiv_id":"2407.01653","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adverse","title":"Deep Reinforcement Learning for Adverse Garage Scenario Generation","date":"2024-07-01","arxiv_id":"2407.01333","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-morphology-towards","title":"Evolutionary Morphology Towards Overconstrained Locomotion via Large-Scale, Multi-Terrain Deep Reinforcement Learning","date":"2024-07-01","arxiv_id":"2407.01050","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-rag-empowered-multi-modal-llm-for","title":"Hybrid RAG-empowered Multi-modal LLM for Secure Data Management in Internet of Medical Things: A Diffusion-based Contract Approach","date":"2024-07-01","arxiv_id":"2407.00978","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-hybrid-a-path-planner-obey-traffic-rules","title":"Let Hybrid A* Path Planner Obey Traffic Rules: A Deep Reinforcement Learning-Based Planning Framework","date":"2024-07-01","arxiv_id":"2407.01216","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalization-and-effective-learning-rates-in","title":"Normalization and effective learning rates in reinforcement learning","date":"2024-07-01","arxiv_id":"2407.01800","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-a-physics-informed-decision","title":"Exploring a Physics-Informed Decision Transformer for Distribution System Restoration: Methodology and Performance Analysis","date":"2024-06-30","arxiv_id":"2407.00808","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-strategies-in","title":"Deep Reinforcement Learning Strategies in Finance: Insights into Asset Holding, Trading Behavior, and Purchase Diversity","date":"2024-06-29","arxiv_id":"2407.09557","repositories_listed":0,"syntology":null},{"url":null,"slug":"tradeoffs-when-considering-deep-reinforcement","title":"Tradeoffs When Considering Deep Reinforcement Learning for Contingency Management in Advanced Air Mobility","date":"2024-06-28","arxiv_id":"2407.00197","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-control-of-a-novel-closed-chain","title":"Autonomous Control of a Novel Closed Chain Five Bar Active Suspension via Deep Reinforcement Learning","date":"2024-06-27","arxiv_id":"2406.18899","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-experts-in-a-mixture-of-rl","title":"Mixture of Experts in a Mixture of RL settings","date":"2024-06-26","arxiv_id":"2406.18420","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-comparison-of-deep-rl-algorithms-1","title":"Performance Comparison of Deep RL Algorithms for Mixed Traffic Cooperative Lane-Changing","date":"2024-06-25","arxiv_id":"2407.02521","repositories_listed":0,"syntology":null},{"url":null,"slug":"fault-detection-for-agents-on-power-grid","title":"Fault Detection for agents on power grid topology optimization: A Comprehensive analysis","date":"2024-06-24","arxiv_id":"2406.16426","repositories_listed":0,"syntology":null},{"url":null,"slug":"cav-ahdv-cav-mitigating-traffic-oscillations","title":"CAV-AHDV-CAV: Mitigating Traffic Oscillations for CAVs through a Novel Car-Following Structure and Reinforcement Learning","date":"2024-06-23","arxiv_id":"2407.02517","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-diagnosing-deep","title":"Understanding and Diagnosing Deep Reinforcement Learning","date":"2024-06-23","arxiv_id":"2406.16979","repositories_listed":0,"syntology":null},{"url":null,"slug":"opticgai-generative-ai-aided-deep","title":"OpticGAI: Generative AI-aided Deep Reinforcement Learning for Optical Networks Optimization","date":"2024-06-22","arxiv_id":"2406.15906","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-uav-path-planning-with-assured","title":"Deep UAV Path Planning with Assured Connectivity in Dense Urban Setting","date":"2024-06-21","arxiv_id":"2406.15225","repositories_listed":0,"syntology":null},{"url":null,"slug":"knobtree-intelligent-database-parameter","title":"KnobTree: Intelligent Database Parameter Configuration via Explainable Reinforcement Learning","date":"2024-06-21","arxiv_id":"2406.15073","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-dynamic-resource-allocation-and","title":"Towards Dynamic Resource Allocation and Client Scheduling in Hierarchical Federated Learning: A Two-Phase Deep Reinforcement Learning Approach","date":"2024-06-21","arxiv_id":"2406.14910","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-neural-networks-for-job-shop-scheduling","title":"Graph Neural Networks for Job Shop Scheduling Problems: A Survey","date":"2024-06-20","arxiv_id":"2406.14096","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-optimization-of-noma-aided-multi-star","title":"Design Optimization of NOMA Aided Multi-STAR-RIS for Indoor Environments: A Convex Approximation Imitated Reinforcement Learning Approach","date":"2024-06-19","arxiv_id":"2406.13280","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-super-human-vision-based-reinforcement","title":"A Super-human Vision-based Reinforcement Learning Agent for Autonomous Racing in Gran Turismo","date":"2024-06-18","arxiv_id":"2406.12563","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-deep-reinforcement-learning-1","title":"Attention-Based Deep Reinforcement Learning for Qubit Allocation in Modular Quantum Architectures","date":"2024-06-17","arxiv_id":"2406.11452","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-uav-multi-ris-qos-aware-aerial","title":"Multi-UAV Multi-RIS QoS-Aware Aerial Communication Systems using DRL and PSO","date":"2024-06-16","arxiv_id":"2406.16934","repositories_listed":0,"syntology":null},{"url":null,"slug":"grad-instructor-universal-backpropagation","title":"Grad-Instructor: Universal Backpropagation with Explainable Evaluation Neural Networks for Meta-learning and AutoML","date":"2024-06-15","arxiv_id":"2406.10559","repositories_listed":0,"syntology":null},{"url":null,"slug":"misam-using-ml-in-dataflow-selection-of","title":"Misam: Using ML in Dataflow Selection of Sparse-Sparse Matrix Multiplication","date":"2024-06-14","arxiv_id":"2406.10166","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-q-learning-for-lane-changing-a","title":"Mix Q-learning for Lane Changing: A Collaborative Decision-Making Method in Multi-Agent Deep Reinforcement Learning","date":"2024-06-14","arxiv_id":"2406.09755","repositories_listed":0,"syntology":null},{"url":null,"slug":"cuer-corrected-uniform-experience-replay-for","title":"CUER: Corrected Uniform Experience Replay for Off-Policy Continuous Deep Reinforcement Learning Algorithms","date":"2024-06-13","arxiv_id":"2406.09030","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-swarm-mesh-refinement-using-deep","title":"Adaptive Swarm Mesh Refinement using Deep Reinforcement Learning with Local Rewards","date":"2024-06-12","arxiv_id":"2406.08440","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-positional","title":"Deep reinforcement learning with positional context for intraday trading","date":"2024-06-12","arxiv_id":"2406.08013","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-go-leveraging-exploration-for","title":"Explore-Go: Leveraging Exploration for Generalisation in Deep Reinforcement Learning","date":"2024-06-12","arxiv_id":"2406.08069","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-deep-reinforcement-learning-for-1","title":"Optimizing Deep Reinforcement Learning for Adaptive Robotic Arm Control","date":"2024-06-12","arxiv_id":"2407.02503","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-training-optimizing-reinforcement","title":"Beyond Training: Optimizing Reinforcement Learning Based Job Shop Scheduling Through Adaptive Action Sampling","date":"2024-06-11","arxiv_id":"2406.07325","repositories_listed":0,"syntology":null},{"url":null,"slug":"dnn-partitioning-task-offloading-and-resource","title":"DNN Partitioning, Task Offloading, and Resource Allocation in Dynamic Vehicular Networks: A Lyapunov-Guided Diffusion-Based Reinforcement Learning Approach","date":"2024-06-11","arxiv_id":"2406.06986","repositories_listed":0,"syntology":null},{"url":null,"slug":"verification-guided-shielding-for-deep","title":"Verification-Guided Shielding for Deep Reinforcement Learning","date":"2024-06-10","arxiv_id":"2406.06507","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-attribute-auction-based-resource","title":"Multi-attribute Auction-based Resource Allocation for Twins Migration in Vehicular Metaverses: A GPT-based DRL Approach","date":"2024-06-08","arxiv_id":"2406.05418","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-policy-distillation-with-decision","title":"Online Policy Distillation with Decision-Attention","date":"2024-06-08","arxiv_id":"2406.05488","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatpcg-large-language-model-driven-reward","title":"ChatPCG: Large Language Model-Driven Reward Design for Procedural Content Generation","date":"2024-06-07","arxiv_id":"2406.11875","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-geological-carbon-storage","title":"Optimization of geological carbon storage operations with multimodal latent dynamic model and deep reinforcement learning","date":"2024-06-07","arxiv_id":"2406.04575","repositories_listed":0,"syntology":null}],"record_sha256":"d2070136244880ee862acebc80473818c323c899282f491cd76159dc8a2bf2b6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}