{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/26","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":26,"pages_in_order":59,"rows_per_page":100,"rows":[2501,2600],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/25","next":"/task/deep-reinforcement-learning/papers/27","papers":[{"url":null,"slug":"uav-enabled-collaborative-beamforming-via","title":"UAV-enabled Collaborative Beamforming via Multi-Agent Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.07453","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-enhanced-coverage-and-capacity","title":"UAV-Assisted Enhanced Coverage and Capacity in Dynamic MU-mMIMO IoT Systems: A Deep Reinforcement Learning Approach","date":"2024-04-10","arxiv_id":"2404.06726","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-approach","title":"Deep Reinforcement Learning-Based Approach for a Single Vehicle Persistent Surveillance Problem with Fuel Constraints","date":"2024-04-09","arxiv_id":"2404.06423","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-braitenberg-style-behaviours-for","title":"Emergent Braitenberg-style Behaviours for Navigating the ViZDoom `My Way Home' Labyrinth","date":"2024-04-09","arxiv_id":"2404.06529","repositories_listed":0,"syntology":null},{"url":null,"slug":"cnn-based-game-state-detection-for-a-foosball","title":"CNN-based Game State Detection for a Foosball Table","date":"2024-04-08","arxiv_id":"2404.05357","repositories_listed":0,"syntology":null},{"url":null,"slug":"computing-transition-pathways-for-the-study","title":"Computing Transition Pathways for the Study of Rare Events Using Deep Reinforcement Learning","date":"2024-04-08","arxiv_id":"2404.05905","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-transformer-for-wireless","title":"Decision Transformers for Wireless Communications: A New Paradigm of Resource Management","date":"2024-04-08","arxiv_id":"2404.05199","repositories_listed":0,"syntology":null},{"url":null,"slug":"ia2-leveraging-instance-aware-index-advisor","title":"IA2: Leveraging Instance-Aware Index Advisor with Reinforcement Learning for Diverse Workloads","date":"2024-04-08","arxiv_id":"2404.05777","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-control-for","title":"Deep Reinforcement Learning Control for Disturbance Rejection in a Nonlinear Dynamic System with Parametric Uncertainty","date":"2024-04-06","arxiv_id":"2404.04699","repositories_listed":0,"syntology":null},{"url":null,"slug":"securing-the-skies-an-irs-assisted-aoi-aware","title":"Securing the Skies: An IRS-Assisted AoI-Aware Secure Multi-UAV System with Efficient Task Offloading","date":"2024-04-06","arxiv_id":"2404.04692","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-response-shaping","title":"Best Response Shaping","date":"2024-04-05","arxiv_id":"2404.06519","repositories_listed":0,"syntology":null},{"url":null,"slug":"intervention-assisted-policy-gradient-methods","title":"Intervention-Assisted Policy Gradient Methods for Online Stochastic Queuing Network Optimization: Technical Report","date":"2024-04-05","arxiv_id":"2404.04106","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-19","title":"A Deep Reinforcement Learning Approach for Security-Aware Service Acquisition in IoT","date":"2024-04-04","arxiv_id":"2404.03276","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-organized-arrival-system-for-urban-air","title":"Self-organized free-flight arrival for urban air mobility","date":"2024-04-04","arxiv_id":"2404.03710","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-traveling","title":"Deep Reinforcement Learning for Traveling Purchaser Problems","date":"2024-04-03","arxiv_id":"2404.02476","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-game-a-model-based-and-imitation","title":"Imitation Game: A Model-based and Imitation Learning Deep Reinforcement Learning Hybrid","date":"2024-04-02","arxiv_id":"2404.01794","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-exploration-all-you-need-effective","title":"Is Exploration All You Need? Effective Exploration Characteristics for Transfer in Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.02235","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-control-camera-exposure-via","title":"Learning to Control Camera Exposure via Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.01636","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-qualitative-and-quantitative-safety","title":"Unifying Qualitative and Quantitative Safety Verification of DNN-Controlled Systems","date":"2024-04-02","arxiv_id":"2404.01769","repositories_listed":0,"syntology":null},{"url":null,"slug":"game-theoretic-deep-reinforcement-learning-to","title":"Game-Theoretic Deep Reinforcement Learning to Minimize Carbon Emissions and Energy Costs for AI Inference Workloads in Geo-Distributed Data Centers","date":"2024-04-01","arxiv_id":"2404.01459","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtlight-efficient-multi-task-reinforcement","title":"MTLight: Efficient Multi-Task Reinforcement Learning for Traffic Signal Control","date":"2024-04-01","arxiv_id":"2404.00886","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-aware-spectrum-pricing-and-power","title":"Privacy-Aware Spectrum Pricing and Power Control Optimization for LEO Satellite Internet-of-Things","date":"2024-04-01","arxiv_id":"2407.00814","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-off-policy-with-model-based","title":"Learning Off-policy with Model-based Intrinsic Motivation For Active Online Exploration","date":"2024-03-31","arxiv_id":"2404.00651","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-mul-multiplier-design-optimization-with","title":"RL-MUL 2.0: Multiplier Design Optimization with Parallel Deep Reinforcement Learning and Space Reduction","date":"2024-03-31","arxiv_id":"2404.00639","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-qap-by-two-stage-graph-pointer","title":"Solving the QAP by Two-Stage Graph Pointer Networks and Reinforcement Learning","date":"2024-03-31","arxiv_id":"2404.00539","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-autoencoders-for-exteroceptive","title":"Variational Autoencoders for exteroceptive perception in reinforcement learning-based collision avoidance","date":"2024-03-31","arxiv_id":"2404.00623","repositories_listed":0,"syntology":null},{"url":null,"slug":"facilitating-reinforcement-learning-for","title":"Facilitating Reinforcement Learning for Process Control Using Transfer Learning: Overview and Perspectives","date":"2024-03-30","arxiv_id":"2404.00247","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learning-based-incentive-mechanism-for","title":"A Learning-based Incentive Mechanism for Mobile AIGC Service in Decentralized Internet of Vehicles","date":"2024-03-29","arxiv_id":"2403.20151","repositories_listed":0,"syntology":null},{"url":null,"slug":"biologically-plausible-topology-improved","title":"Biologically-Plausible Topology Improved Spiking Actor Network for Efficient Deep Reinforcement Learning","date":"2024-03-29","arxiv_id":"2403.20163","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-planning-of-magnetic-microswimmers-in","title":"Optimal navigation of magnetic artificial microswimmers in blood capillaries with deep reinforcement learning","date":"2024-03-29","arxiv_id":"2404.02171","repositories_listed":0,"syntology":null},{"url":null,"slug":"removing-the-need-for-ground-truth-uwb-data","title":"Removing the need for ground truth UWB data collection: self-supervised ranging error correction using deep reinforcement learning","date":"2024-03-28","arxiv_id":"2403.19262","repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-constraints-as-terminations-for-legged","title":"CaT: Constraints as Terminations for Legged Locomotion Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18765","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-traffic-signal-control-via-genetic","title":"Learning Traffic Signal Control via Genetic Programming","date":"2024-03-26","arxiv_id":"2403.17328","repositories_listed":0,"syntology":null},{"url":null,"slug":"vdsc-enhancing-exploration-timing-with-value","title":"VDSC: Enhancing Exploration Timing with Value Discrepancy and State Counts","date":"2024-03-26","arxiv_id":"2403.17542","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and-mean-variance","title":"Deep Reinforcement Learning and Mean-Variance Strategies for Responsible Portfolio Optimization","date":"2024-03-25","arxiv_id":"2403.16667","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-level-control-of-weaving-sections-in-mixed","title":"Bi-Level Control of Weaving Sections in Mixed Traffic Environments with Connected and Automated Vehicles","date":"2024-03-24","arxiv_id":"2403.16225","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-modeling-of-deep-reinforcement","title":"Interpretable Modeling of Deep Reinforcement Learning Driven Scheduling","date":"2024-03-24","arxiv_id":"2403.16293","repositories_listed":0,"syntology":null},{"url":null,"slug":"blockchain-based-pseudonym-management-for","title":"Blockchain-based Pseudonym Management for Vehicle Twin Migrations in Vehicular Edge Metaverse","date":"2024-03-22","arxiv_id":"2403.15285","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-long-short-term-memory-based","title":"Improved Long Short-Term Memory-based Wastewater Treatment Simulators for Deep Reinforcement Learning","date":"2024-03-22","arxiv_id":"2403.15091","repositories_listed":0,"syntology":null},{"url":null,"slug":"srlm-human-in-loop-interactive-social-robot","title":"Unifying Large Language Model and Deep Reinforcement Learning for Human-in-Loop Interactive Socially-aware Navigation","date":"2024-03-22","arxiv_id":"2403.15648","repositories_listed":0,"syntology":null},{"url":null,"slug":"dourn-improving-douzero-by-residual-neural","title":"DouRN: Improving DouZero by Residual Neural Networks","date":"2024-03-21","arxiv_id":"2403.14102","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-value-tracking-for-deep-reinforcement","title":"Fast Value Tracking for Deep Reinforcement Learning","date":"2024-03-19","arxiv_id":"2403.13178","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-deep-reinforcement-learning","title":"Demystifying the Physics of Deep Reinforcement Learning-Based Autonomous Vehicle Decision-Making","date":"2024-03-18","arxiv_id":"2403.11432","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-crossover","title":"Deep Neural Crossover","date":"2024-03-17","arxiv_id":"2403.11159","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-drone-and-mobile-charger-via","title":"Scheduling Drone and Mobile Charger via Hybrid-Action Deep Reinforcement Learning","date":"2024-03-16","arxiv_id":"2403.10761","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-neural-network-with-region-of","title":"Lyapunov Neural Network with Region of Attraction Search","date":"2024-03-15","arxiv_id":"2403.10621","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-constraint-programming-in-a-deep","title":"Leveraging Constraint Programming in a Deep Learning Approach for Dynamically Solving the Flexible Job-Shop Scheduling Problem","date":"2024-03-14","arxiv_id":"2403.09249","repositories_listed":0,"syntology":null},{"url":null,"slug":"socially-integrated-navigation-a-social","title":"Socially Integrated Navigation: A Social Acting Robot with Deep Reinforcement Learning","date":"2024-03-14","arxiv_id":"2403.09793","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-assisted-reinforcement-learning","title":"Digital Twin-assisted Reinforcement Learning for Resource-aware Microservice Offloading in Edge Computing","date":"2024-03-13","arxiv_id":"2403.08687","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-operators-for-enabling-parallel-planning","title":"Meta-operators for Enabling Parallel Planning Using Deep Reinforcement Learning","date":"2024-03-13","arxiv_id":"2403.08910","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-strategy-for-blood-glucose","title":"An Improved Strategy for Blood Glucose Control Using Multi-Step Deep Reinforcement Learning","date":"2024-03-12","arxiv_id":"2403.07566","repositories_listed":0,"syntology":null},{"url":null,"slug":"symmetric-q-learning-reducing-skewness-of","title":"Symmetric Q-learning: Reducing Skewness of Bellman Error in Online Reinforcement Learning","date":"2024-03-12","arxiv_id":"2403.07704","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-modelling","title":"Deep Reinforcement Learning for Modelling Protein Complexes","date":"2024-03-11","arxiv_id":"2405.02299","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactical-decision-making-for-autonomous","title":"Tactical Decision Making for Autonomous Trucks by Deep Reinforcement Learning with Total Cost of Operation Based Reward","date":"2024-03-11","arxiv_id":"2403.06524","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-policy-sparsification-and-low-rank","title":"Optimal Policy Sparsification and Low Rank Decomposition for Deep Reinforcement Learning","date":"2024-03-10","arxiv_id":"2403.06313","repositories_listed":0,"syntology":null},{"url":null,"slug":"dissecting-deep-rl-with-high-update-ratios","title":"Dissecting Deep RL with High Update Ratios: Combatting Value Divergence","date":"2024-03-09","arxiv_id":"2403.05996","repositories_listed":0,"syntology":null},{"url":null,"slug":"udcr-unsupervised-aortic-dsa-cta-rigid","title":"UDCR: Unsupervised Aortic DSA/CTA Rigid Registration Using Deep Reinforcement Learning and Overlap Degree Calculation","date":"2024-03-09","arxiv_id":"2403.05753","repositories_listed":0,"syntology":null},{"url":null,"slug":"shielded-deep-reinforcement-learning-for","title":"Shielded Deep Reinforcement Learning for Complex Spacecraft Tasking","date":"2024-03-08","arxiv_id":"2403.05693","repositories_listed":0,"syntology":null},{"url":null,"slug":"fill-and-spill-deep-reinforcement-learning","title":"Fill-and-Spill: Deep Reinforcement Learning Policy Gradient Methods for Reservoir Operation Decision and Control","date":"2024-03-07","arxiv_id":"2403.04195","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-cooperative-eco-driving-via","title":"Generalizing Cooperative Eco-driving via Multi-residual Task Learning","date":"2024-03-07","arxiv_id":"2403.04232","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-load-frequency-control-of","title":"Model-Free Load Frequency Control of Nonlinear Power Systems Based on Deep Reinforcement Learning","date":"2024-03-07","arxiv_id":"2403.04374","repositories_listed":0,"syntology":null},{"url":null,"slug":"noisy-spiking-actor-network-for-exploration","title":"Noisy Spiking Actor Network for Exploration","date":"2024-03-07","arxiv_id":"2403.04162","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlearn-off-policy-learning-with-efficient","title":"Vlearn: Off-Policy Learning with Efficient State-Value Function Estimation","date":"2024-03-07","arxiv_id":"2403.04453","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexterous-legged-locomotion-in-confined-3d","title":"Dexterous Legged Locomotion in Confined 3D Spaces with Reinforcement Learning","date":"2024-03-06","arxiv_id":"2403.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"population-aware-online-mirror-descent-for","title":"Population-aware Online Mirror Descent for Mean-Field Games by Deep Reinforcement Learning","date":"2024-03-06","arxiv_id":"2403.03552","repositories_listed":0,"syntology":null},{"url":null,"slug":"stop-regressing-training-value-functions-via","title":"Stop Regressing: Training Value Functions via Classification for Scalable Deep RL","date":"2024-03-06","arxiv_id":"2403.03950","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-zero-shot-reinforcement-learning-strategy","title":"A Zero-Shot Reinforcement Learning Strategy for Autonomous Guidewire Navigation","date":"2024-03-05","arxiv_id":"2403.02777","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-vehicle-decision-and-control","title":"Autonomous vehicle decision and control through reinforcement learning with traffic flow randomization","date":"2024-03-05","arxiv_id":"2403.02882","repositories_listed":0,"syntology":null},{"url":null,"slug":"fighting-game-adaptive-background-music-for","title":"Fighting Game Adaptive Background Music for Improved Gameplay","date":"2024-03-05","arxiv_id":"2403.02701","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-federated-learning-and-edge","title":"Leveraging Federated Learning and Edge Computing for Recommendation Systems within Cloud Computing Networks","date":"2024-03-05","arxiv_id":"2403.03165","repositories_listed":0,"syntology":null},{"url":null,"slug":"race-sm-reinforcement-learning-based","title":"RACE-SM: Reinforcement Learning Based Autonomous Control for Social On-Ramp Merging","date":"2024-03-05","arxiv_id":"2403.03359","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-policy-transfer-with-effect","title":"Cross Domain Policy Transfer with Effect Cycle-Consistency","date":"2024-03-04","arxiv_id":"2403.02018","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dynamic-5","title":"Deep Reinforcement Learning for Dynamic Algorithm Selection: A Proof-of-Principle Study on Differential Evolution","date":"2024-03-04","arxiv_id":"2403.02131","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterated-q-network-beyond-the-one-step","title":"Iterated $Q$-Network: Beyond One-Step Bellman Updates in Deep Reinforcement Learning","date":"2024-03-04","arxiv_id":"2403.02107","repositories_listed":0,"syntology":null},{"url":null,"slug":"twisting-lids-off-with-two-hands","title":"Twisting Lids Off with Two Hands","date":"2024-03-04","arxiv_id":"2403.02338","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloud-based-federated-learning-framework-for","title":"Cloud-based Federated Learning Framework for MRI Segmentation","date":"2024-03-01","arxiv_id":"2403.00254","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-solving","title":"Deep Reinforcement Learning for Solving Management Problems: Towards A Large Management Mode","date":"2024-03-01","arxiv_id":"2403.00318","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-deep-reinforcement-learning-through-3","title":"Robust Deep Reinforcement Learning Through Adversarial Attacks and Training : A Survey","date":"2024-03-01","arxiv_id":"2403.00420","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustifying-a-policy-in-multi-agent-rl-with","title":"Robustifying a Policy in Multi-Agent RL with Diverse Cooperative Behaviors and Adversarial Style Sampling for Assistive Tasks","date":"2024-03-01","arxiv_id":"2403.00344","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-a-convex","title":"Deep Reinforcement Learning: A Convex Optimization Approach","date":"2024-02-29","arxiv_id":"2402.19212","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-the-causes-of-plasticity-loss","title":"Disentangling the Causes of Plasticity Loss in Neural Networks","date":"2024-02-29","arxiv_id":"2402.18762","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-aware-deep-reinforcement-learning-1","title":"Temporal-Aware Deep Reinforcement Learning for Energy Storage Bidding in Energy and Contingency Reserve Markets","date":"2024-02-29","arxiv_id":"2402.19110","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-offloading-in-semantic-aware","title":"Computational Offloading in Semantic-Aware Cloud-Edge-End Collaborative Networks","date":"2024-02-28","arxiv_id":"2402.18183","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-fusion-of-deep-reinforcement-learning-and","title":"The Fusion of Deep Reinforcement Learning and Edge Computing for Real-time Monitoring and Control Optimization in IoT Environments","date":"2024-02-28","arxiv_id":"2403.07923","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-do-animals-need-shaping-a-theory-of-task","title":"Why Do Animals Need Shaping? A Theory of Task Composition and Curriculum Learning","date":"2024-02-28","arxiv_id":"2402.18361","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-investment-frontiers-industry-grade","title":"Advancing Investment Frontiers: Industry-grade Deep Reinforcement Learning for Portfolio Optimization","date":"2024-02-27","arxiv_id":"2403.07916","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergency-caching-coded-caching-based","title":"Emergency Caching: Coded Caching-based Reliable Map Transmission in Emergency Networks","date":"2024-02-27","arxiv_id":"2402.17550","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-deep-deterministic-policy-gradient","title":"Model Free Deep Deterministic Policy Gradient Controller for Setpoint Tracking of Non-minimum Phase Systems","date":"2024-02-27","arxiv_id":"2402.17703","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-16","title":"Multi-Agent Deep Reinforcement Learning for Distributed Satellite Routing","date":"2024-02-27","arxiv_id":"2402.17666","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-communication-enabled-wireless","title":"Rate Splitting Multiple Access-Enabled Adaptive Panoramic Video Semantic Transmission","date":"2024-02-26","arxiv_id":"2402.16581","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-prior-estimates-for-deep-residual-network","title":"A priori Estimates for Deep Residual Network in Continuous-time Reinforcement Learning","date":"2024-02-24","arxiv_id":"2402.16899","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-transformer-based-deep","title":"Combining Transformer based Deep Reinforcement Learning with Black-Litterman Model for Portfolio Optimization","date":"2024-02-23","arxiv_id":"2402.16609","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-circuit-optimization-with-alphatensor","title":"Quantum Circuit Optimization with AlphaTensor","date":"2024-02-22","arxiv_id":"2402.14396","repositories_listed":0,"syntology":null},{"url":null,"slug":"shm-traffic-drl-and-transfer-learning-based","title":"SHM-Traffic: DRL and Transfer learning based UAV Control for Structural Health Monitoring of Bridges with Traffic","date":"2024-02-22","arxiv_id":"2402.14757","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-the-game-of-guandan-with-deep","title":"Mastering the Game of Guandan with Deep Reinforcement Learning and Behavior Regulating","date":"2024-02-21","arxiv_id":"2402.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesis-of-hierarchical-controllers-based","title":"Synthesis of Hierarchical Controllers Based on Deep Reinforcement Learning Policies","date":"2024-02-21","arxiv_id":"2402.13785","repositories_listed":0,"syntology":null},{"url":null,"slug":"antifragile-perimeter-control-anticipating","title":"Antifragile Perimeter Control: Anticipating and Gaining from Disruptions with Reinforcement Learning","date":"2024-02-20","arxiv_id":"2402.12665","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hedging-with-market-impact","title":"Deep Hedging with Market Impact","date":"2024-02-20","arxiv_id":"2402.13326","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-behavioral-modes-in-deep","title":"Discovering Behavioral Modes in Deep Reinforcement Learning Policies Using Trajectory Clustering in Latent Space","date":"2024-02-20","arxiv_id":"2402.12939","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-framework-for-adaptive-stress-testing","title":"A novel framework for adaptive stress testing of autonomous vehicles in highways","date":"2024-02-19","arxiv_id":"2402.11813","repositories_listed":0,"syntology":null}],"record_sha256":"7cae7696a11296182b683f78f6e1f8a6953067cd89ceb5975d4a46b17d4a0a3b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}