{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/41","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":41,"pages_in_order":59,"rows_per_page":100,"rows":[4001,4100],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/40","next":"/task/deep-reinforcement-learning/papers/42","papers":[{"url":null,"slug":"multi-agent-deep-reinforcement-learning-madrl","title":"Multi-agent deep reinforcement learning (MADRL) meets multi-user MIMO systems","date":"2021-09-10","arxiv_id":"2109.04986","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-a-domestic-battery-and-solar","title":"Optimizing a domestic battery and solar photovoltaic system with deep reinforcement learning","date":"2021-09-10","arxiv_id":"2109.05024","repositories_listed":0,"syntology":null},{"url":null,"slug":"dan-decentralized-attention-based-neural","title":"DAN: Decentralized Attention-based Neural Network for the MinMax Multiple Traveling Salesman Problem","date":"2021-09-09","arxiv_id":"2109.04205","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-equal-risk","title":"Deep Reinforcement Learning for Equal Risk Pricing and Hedging under Dynamic Expectile Risk Measures","date":"2021-09-09","arxiv_id":"2109.04001","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-8","title":"A Deep Reinforcement Learning Approach for Online Parcel Assignment","date":"2021-09-08","arxiv_id":"2109.03467","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-deep-reinforcement-learning-in-1","title":"A Survey of Deep Reinforcement Learning in Recommender Systems: A Systematic Review and Future Directions","date":"2021-09-08","arxiv_id":"2109.03540","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-did-you-learn-that-from-surprising","title":"Membership Inference Attacks Against Temporally Correlated Data in Deep Reinforcement Learning","date":"2021-09-08","arxiv_id":"2109.03975","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindsight-reward-tweaking-via-conditional","title":"Hindsight Reward Tweaking via Conditional Deep Reinforcement Learning","date":"2021-09-06","arxiv_id":"2109.02332","repositories_listed":0,"syntology":null},{"url":null,"slug":"soft-hierarchical-graph-recurrent-networks","title":"Soft Hierarchical Graph Recurrent Networks for Many-Agent Partially Observable Environments","date":"2021-09-05","arxiv_id":"2109.02032","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-battery-energy","title":"Reinforcement Learning for Battery Energy Storage Dispatch augmented with Model-based Optimizer","date":"2021-09-02","arxiv_id":"2109.01659","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-computation-efficient-device","title":"Communication-Computation Efficient Device-Edge Co-Inference via AutoML","date":"2021-08-30","arxiv_id":"2108.13009","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-vulnerabilities-of-deep-neural","title":"Investigating Vulnerabilities of Deep Neural Policies","date":"2021-08-30","arxiv_id":"2108.13093","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-planning-for-cellular-connected-uav-a","title":"Path Planning for Cellular-Connected UAV: A DRL Solution with Quantum-Inspired Experience Replay","date":"2021-08-30","arxiv_id":"2108.13184","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-policy-efficient-reduction-approach-to","title":"A Policy Efficient Reduction Approach to Convex Constrained Deep Reinforcement Learning","date":"2021-08-29","arxiv_id":"2108.12916","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-curiosity-for-real-time-training","title":"Autonomous Curiosity for Real-Time Training Onboard Robotic Agents","date":"2021-08-29","arxiv_id":"2109.00927","repositories_listed":0,"syntology":null},{"url":null,"slug":"harvesting-idle-resources-in-serverless","title":"Accelerating Serverless Computing by Harvesting Idle Resources","date":"2021-08-28","arxiv_id":"2108.12717","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-wireless-2","title":"Deep Reinforcement Learning for Wireless Resource Allocation Using Buffer State Information","date":"2021-08-27","arxiv_id":"2108.12198","repositories_listed":0,"syntology":null},{"url":null,"slug":"wad-a-deep-reinforcement-learning-agent-for","title":"WAD: A Deep Reinforcement Learning Agent for Urban Autonomous Driving","date":"2021-08-27","arxiv_id":"2108.12134","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dynamic-band","title":"Deep Reinforcement Learning for Dynamic Band Switch in Cellular-Connected UAV","date":"2021-08-26","arxiv_id":"2108.12054","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-computer","title":"Deep Reinforcement Learning in Computer Vision: A Comprehensive Survey","date":"2021-08-25","arxiv_id":"2108.11510","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-aware-model-initialization-for","title":"Entropy-Aware Model Initialization for Effective Exploration in Deep Reinforcement Learning","date":"2021-08-24","arxiv_id":"2108.10533","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-dba-no-regret-multi-armed-bandits-for","title":"No DBA? No regret! Multi-armed bandits for index tuning of analytical and HTAP workloads with provable guarantees","date":"2021-08-23","arxiv_id":"2108.10130","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-gnss","title":"A Reinforcement Learning Approach for GNSS Spoofing Attack Detection of Autonomous Vehicles","date":"2021-08-19","arxiv_id":"2108.08628","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-difficulty-adjustment-in-virtual","title":"Dynamic Difficulty Adjustment in Virtual Reality Exergames through Experience-driven Procedural Content Generation","date":"2021-08-19","arxiv_id":"2108.08762","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-deep-reinforcement-learning-using","title":"Explainable Deep Reinforcement Learning Using Introspection in a Non-episodic Task","date":"2021-08-18","arxiv_id":"2108.08911","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-placement-of-public-electric-vehicle","title":"Optimal Placement of Public Electric Vehicle Charging Stations Using Deep Reinforcement Learning","date":"2021-08-17","arxiv_id":"2108.07772","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-to-tree-policy-distillation-with","title":"Neural-to-Tree Policy Distillation with Policy Improvement Criterion","date":"2021-08-16","arxiv_id":"2108.06898","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-scheduling-of-isolated-microgrids","title":"Optimal Scheduling of Isolated Microgrids Using Automated Reinforcement Learning-based Multi-period Forecasting","date":"2021-08-15","arxiv_id":"2108.06764","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-microscopic-pandemic-simulator-for-pandemic","title":"A Microscopic Pandemic Simulator for Pandemic Prediction Using Scalable Million-Agent Reinforcement Learning","date":"2021-08-14","arxiv_id":"2108.06589","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-explicit-prediction-matter-in-energy","title":"Does Explicit Prediction Matter in Deep Reinforcement Learning-Based Energy Management?","date":"2021-08-11","arxiv_id":"2108.05099","repositories_listed":0,"syntology":null},{"url":null,"slug":"dq-gat-towards-safe-and-efficient-autonomous","title":"DQ-GAT: Towards Safe and Efficient Autonomous Driving with Deep Q-Learning and Graph Attention Networks","date":"2021-08-11","arxiv_id":"2108.05030","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-deep-reinforcement-learning-for-1","title":"A Survey on Deep Reinforcement Learning for Data Processing and Analytics","date":"2021-08-10","arxiv_id":"2108.04526","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-demand-driven","title":"Deep Reinforcement Learning for Demand Driven Services in Logistics and Transportation Systems: A Survey","date":"2021-08-10","arxiv_id":"2108.04462","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-related-search-query-suggestions","title":"High Quality Related Search Query Suggestions using Deep Reinforcement Learning","date":"2021-08-10","arxiv_id":"2108.04452","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-dense-and-sparse-visual-rewards-in","title":"A Study on Dense and Sparse (Visual) Rewards in Robot Policy Learning","date":"2021-08-06","arxiv_id":"2108.03222","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-intelligent-3","title":"Deep Reinforcement Learning for Intelligent Reflecting Surface-assisted D2D Communications","date":"2021-08-06","arxiv_id":"2108.02892","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-continuous","title":"Deep Reinforcement Learning for Continuous Docking Control of Autonomous Underwater Vehicles: A Benchmarking Study","date":"2021-08-05","arxiv_id":"2108.02665","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-slice-placement-under-non","title":"DRL-based Slice Placement Under Non-Stationary Conditions","date":"2021-08-05","arxiv_id":"2108.02495","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-consensus-based-representation-deep","title":"Ensemble Consensus-based Representation Deep Reinforcement Learning for Hybrid FSO/RF Communication Systems","date":"2021-08-05","arxiv_id":"2108.02551","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-design-and-construct-bridge","title":"Learning to Design and Construct Bridge without Blueprint","date":"2021-08-05","arxiv_id":"2108.02439","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-robustness-of-controlled-deep","title":"On the Robustness of Controlled Deep Reinforcement Learning for Slice Placement","date":"2021-08-05","arxiv_id":"2108.02505","repositories_listed":0,"syntology":null},{"url":null,"slug":"ris-assisted-uav-communications-for-iot-with","title":"RIS-assisted UAV Communications for IoT with Wireless Power Transfer Using Deep Reinforcement Learning","date":"2021-08-05","arxiv_id":"2108.02889","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-performance-across-two-atari-paddle","title":"High Performance Across Two Atari Paddle Games Using the Same Perceptual Control Architecture Without Training","date":"2021-08-04","arxiv_id":"2108.01895","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlled-deep-reinforcement-learning-for","title":"Controlled Deep Reinforcement Learning for Optimized Slice Placement","date":"2021-08-03","arxiv_id":"2108.01544","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-networked","title":"Deep Reinforcement Learning Based Networked Control with Network Delays for Signal Temporal Logic Specifications","date":"2021-08-03","arxiv_id":"2108.01317","repositories_listed":0,"syntology":null},{"url":null,"slug":"factor-representation-and-decision-making-in","title":"Factor Representation and Decision Making in Stock Markets Using Deep Reinforcement Learning","date":"2021-08-03","arxiv_id":"2108.01758","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-against-deep","title":"Adversarial Attacks Against Deep Reinforcement Learning Framework in Internet of Vehicles","date":"2021-08-02","arxiv_id":"2108.00833","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-5","title":"A Reinforcement Learning Approach for Scheduling in mmWave Networks","date":"2021-08-01","arxiv_id":"2108.00548","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-reinforcement-learning-for-table","title":"Interactive Reinforcement Learning for Table Balancing Robot","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-trajectory-planning-in-wireless-sensor","title":"UAV Trajectory Planning in Wireless Sensor Networks for Energy Consumption Minimization by Deep Reinforcement Learning","date":"2021-08-01","arxiv_id":"2108.00354","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-intelligent-energy-management-framework","title":"An Intelligent Energy Management Framework for Hybrid-Electric Propulsion Systems Using Deep Reinforcement Learning","date":"2021-07-31","arxiv_id":"2108.00256","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-dyna-style-model-based-deep","title":"Physics-informed Dyna-Style Model-Based Deep Reinforcement Learning for Dynamic Control","date":"2021-07-31","arxiv_id":"2108.00128","repositories_listed":0,"syntology":null},{"url":null,"slug":"snippet-policy-network-for-multi-class-varied","title":"Snippet Policy Network for Multi-class Varied-length ECG Early Classification","date":"2021-07-28","arxiv_id":"2107.13361","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-reinforcement-learning-for","title":"Value-Based Reinforcement Learning for Continuous Control Robotic Manipulation in Multi-Task Sparse Reward Settings","date":"2021-07-28","arxiv_id":"2107.13356","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-l3-slice","title":"Deep Reinforcement Learning for L3 Slice Localization in Sarcopenia Assessment","date":"2021-07-27","arxiv_id":"2107.12800","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-deep-q-learning-based-real-time","title":"Double Deep Q-learning Based Real-Time Optimization Strategy for Microgrids","date":"2021-07-27","arxiv_id":"2107.12545","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-prediction-rendering-and-1","title":"Learning-based Prediction, Rendering and Transmission for Interactive Virtual Reality in RIS-Assisted Terahertz Networks","date":"2021-07-27","arxiv_id":"2107.12943","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-game-engagement-and-difficulty","title":"Predicting Game Engagement and Difficulty Using AI Players","date":"2021-07-26","arxiv_id":"2107.12061","repositories_listed":0,"syntology":null},{"url":null,"slug":"dr2l-surfacing-corner-cases-to-robustify","title":"DR2L: Surfacing Corner Cases to Robustify Autonomous Driving via Domain Randomization Reinforcement Learning","date":"2021-07-25","arxiv_id":"2107.11762","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-exploration-for-multi-agent-deep","title":"Cooperative Exploration for Multi-Agent Deep Reinforcement Learning","date":"2021-07-23","arxiv_id":"2107.11444","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-equal-risk-pricing-of-financial-1","title":"Deep equal risk pricing of financial derivatives with non-translation invariant risk measures","date":"2021-07-23","arxiv_id":"2107.11340","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-7","title":"A Deep Reinforcement Learning Approach for Fair Traffic Signal Control","date":"2021-07-21","arxiv_id":"2107.10146","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-controller-fusion-leveraging-control","title":"Bayesian Controller Fusion: Leveraging Control Priors in Deep Reinforcement Learning for Robotics","date":"2021-07-21","arxiv_id":"2107.09822","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-reward-shaping-for-efficient","title":"Multimodal Reward Shaping for Efficient Exploration in Reinforcement Learning","date":"2021-07-19","arxiv_id":"2107.08888","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-accuracy-model-based-reinforcement","title":"High-Accuracy Model-Based Reinforcement Learning, a Survey","date":"2021-07-17","arxiv_id":"2107.08241","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-robustness-of-deep-reinforcement","title":"On the Robustness of Deep Reinforcement Learning in IRS-Aided Wireless Communications Systems","date":"2021-07-17","arxiv_id":"2107.08293","repositories_listed":0,"syntology":null},{"url":null,"slug":"modrl-d-el-multiobjective-deep-reinforcement","title":"MODRL/D-EL: Multiobjective Deep Reinforcement Learning with Evolutionary Learning for Multiobjective Optimization","date":"2021-07-16","arxiv_id":"2107.07961","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-adversarial-imitation-learning-using","title":"Visual Adversarial Imitation Learning using Variational Models","date":"2021-07-16","arxiv_id":"2107.08829","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-dynamic-2","title":"Deep Reinforcement Learning based Dynamic Optimization of Bus Timetable","date":"2021-07-15","arxiv_id":"2107.07066","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-human-ai-teams-for-learned-and","title":"Evaluation of Human-AI Teams for Learned and Rule-Based Agents in Hanabi","date":"2021-07-15","arxiv_id":"2107.07630","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-embedded-multi-agent-learning-for-smart","title":"Graph-Embedded Multi-Agent Learning for Smart Reconfigurable THz MIMO-NOMA Networks","date":"2021-07-15","arxiv_id":"2107.07198","repositories_listed":0,"syntology":null},{"url":null,"slug":"going-beyond-linear-rl-sample-efficient","title":"Going Beyond Linear RL: Sample Efficient Neural Function Approximation","date":"2021-07-14","arxiv_id":"2107.06466","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixing-human-demonstrations-with-self","title":"Mixing Human Demonstrations with Self-Exploration in Experience Replay for Deep Reinforcement Learning","date":"2021-07-14","arxiv_id":"2107.06840","repositories_listed":0,"syntology":null},{"url":null,"slug":"qos-aware-scheduling-in-new-radio-using-deep","title":"QoS-Aware Scheduling in New Radio Using Deep Reinforcement Learning","date":"2021-07-14","arxiv_id":"2107.06570","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-6","title":"A Deep Reinforcement Learning Approach for Traffic Signal Control Optimization","date":"2021-07-13","arxiv_id":"2107.06115","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-e2e-energy-efficient-in-joint","title":"Learning based E2E Energy Efficient in Joint Radio and NFV Resource Allocation for 5G and Beyond Networks","date":"2021-07-13","arxiv_id":"2107.05991","repositories_listed":0,"syntology":null},{"url":"/paper/r3l-connecting-deep-reinforcement-learning-to","slug":"r3l-connecting-deep-reinforcement-learning-to","title":"R3L: Connecting Deep Reinforcement Learning to Recurrent Neural Networks for Image Denoising via Residual Recovery","date":"2021-07-12","arxiv_id":"2107.05318","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for-4","title":"Distributed Deep Reinforcement Learning for Intelligent Traffic Monitoring with a Team of Aerial Robots","date":"2021-07-10","arxiv_id":"2107.04924","repositories_listed":0,"syntology":null},{"url":null,"slug":"attend2pack-bin-packing-through-deep","title":"Attend2Pack: Bin Packing through Deep Reinforcement Learning with Attention","date":"2021-07-09","arxiv_id":"2107.04333","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-gain-control-through-deep","title":"Automated Gain Control Through Deep Reinforcement Learning for Downstream Radar Object Detection","date":"2021-07-08","arxiv_id":"2107.03792","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-autonomous-pipeline-inspection-with","title":"Towards Autonomous Pipeline Inspection with Hierarchical Reinforcement Learning","date":"2021-07-08","arxiv_id":"2107.03685","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-progress-of-deep-reinforcement","title":"Evaluating the progress of Deep Reinforcement Learning in the real world: aligning domain-agnostic and domain-specific research","date":"2021-07-07","arxiv_id":"2107.03015","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-model-free-hedging-for-variable","title":"Pseudo-Model-Free Hedging for Variable Annuities via Deep Reinforcement Learning","date":"2021-07-07","arxiv_id":"2107.03340","repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-smart-traffic-signal-control-on","title":"Effects of Smart Traffic Signal Control on Air Quality","date":"2021-07-06","arxiv_id":"2107.02361","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-rough-terrain-vehicles-using-deep","title":"Control of rough terrain vehicles using deep reinforcement learning","date":"2021-07-05","arxiv_id":"2107.01867","repositories_listed":0,"syntology":null},{"url":null,"slug":"winning-at-any-cost-infringing-the-cartel","title":"Winning at Any Cost -- Infringing the Cartel Prohibition With Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01856","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dimensional-state-and-action","title":"Low-Dimensional State and Action Representation Learning with MDP Homomorphism Metrics","date":"2021-07-04","arxiv_id":"2107.01677","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-restless-bandits-tackling-interval","title":"Restless and Uncertain: Robust Policies for Restless Bandits via Deep Multi-Agent Reinforcement Learning","date":"2021-07-04","arxiv_id":"2107.01689","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-signal-control-with-communicative","title":"Traffic Signal Control with Communicative Deep Reinforcement Learning Agents: a Case Study","date":"2021-07-03","arxiv_id":"2107.01347","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-deep-reinforcement-learning-based","title":"A Novel Deep Reinforcement Learning Based Stock Direction Prediction using Knowledge Graph and Community Aware Sentiments","date":"2021-07-02","arxiv_id":"2107.00931","repositories_listed":0,"syntology":null},{"url":null,"slug":"socialai-benchmarking-socio-cognitive","title":"SocialAI: Benchmarking Socio-Cognitive Abilities in Deep Reinforcement Learning Agents","date":"2021-07-02","arxiv_id":"2107.00956","repositories_listed":0,"syntology":null},{"url":null,"slug":"drone-swarm-patrolling-with-uneven-coverage","title":"Drone swarm patrolling with uneven coverage requirements","date":"2021-07-01","arxiv_id":"2107.00362","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-power-allocation-for-rate-splitting","title":"Optimal Power Allocation for Rate Splitting Communications with Deep Reinforcement Learning","date":"2021-07-01","arxiv_id":"2107.00238","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-the-free-energy-principle-to","title":"Applications of the Free Energy Principle to Machine Learning and Neuroscience","date":"2021-06-30","arxiv_id":"2107.00140","repositories_listed":0,"syntology":null},{"url":null,"slug":"drill-deep-reinforcement-learning-for","title":"DRILL-- Deep Reinforcement Learning for Refinement Operators in $\\mathcal{ALC}$","date":"2021-06-29","arxiv_id":"2106.15373","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-online-machine-learning-over","title":"UAV-assisted Online Machine Learning over Multi-Tiered Networks: A Hierarchical Nested Personalized Federated Learning Approach","date":"2021-06-29","arxiv_id":"2106.15734","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-q-learning-deep-q-learning-with-state","title":"Expert Q-learning: Deep Reinforcement Learning with Coarse State Values from Offline Expert Examples","date":"2021-06-28","arxiv_id":"2106.14642","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-with-deep-reinforcement-1","title":"Continuous Control with Deep Reinforcement Learning for Autonomous Vessels","date":"2021-06-27","arxiv_id":"2106.14130","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-robot-deep-reinforcement-learning-for","title":"Hierarchically Integrated Models: Learning to Navigate from Heterogeneous Robots","date":"2021-06-24","arxiv_id":"2106.13280","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-reinforcement-learning-with","title":"Off-Policy Reinforcement Learning with Delayed Rewards","date":"2021-06-22","arxiv_id":"2106.11854","repositories_listed":0,"syntology":null}],"record_sha256":"3c4ad2f2ec56684caf9adecab9c47d4dbde9c41433f56972a3709368b8d6df09","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}