{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/48","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":48,"pages_in_order":59,"rows_per_page":100,"rows":[4701,4800],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/47","next":"/task/deep-reinforcement-learning/papers/49","papers":[{"url":null,"slug":"representations-for-stable-off-policy","title":"Representations for Stable Off-Policy Reinforcement Learning","date":"2020-07-10","arxiv_id":"2007.05520","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-prune-deep-neural-networks-via-2","title":"Learning to Prune Deep Neural Networks via Reinforcement Learning","date":"2020-07-09","arxiv_id":"2007.04756","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dose-ct-denoising-via-joint-bilateral","title":"Low Dose CT Denoising via Joint Bilateral Filtering and Intelligent Parameter Optimization","date":"2020-07-09","arxiv_id":"2007.04768","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-model-based-on","title":"A deep reinforcement learning model based on deterministic policy gradient for collective neural crest cell migration","date":"2020-07-07","arxiv_id":"2007.03190","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognitive-radio-network-throughput","title":"Cognitive Radio Network Throughput Maximization with Deep Reinforcement Learning","date":"2020-07-07","arxiv_id":"2007.03165","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and-its","title":"Deep Reinforcement Learning and its Neuroscientific Implications","date":"2020-07-07","arxiv_id":"2007.03750","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-interactive","title":"Deep Reinforcement Learning with Interactive Feedback in a Human-Robot Environment","date":"2020-07-07","arxiv_id":"2007.03363","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-maintenance-for-edge-based-sensor","title":"Predictive Maintenance for Edge-Based Sensor Networks: A Deep Reinforcement Learning Approach","date":"2020-07-07","arxiv_id":"2007.03313","repositories_listed":0,"syntology":null},{"url":null,"slug":"consensus-multi-agent-reinforcement-learning","title":"Consensus Multi-Agent Reinforcement Learning for Volt-VAR Control in Power Distribution Networks","date":"2020-07-06","arxiv_id":"2007.02991","repositories_listed":0,"syntology":null},{"url":null,"slug":"robo-gym-an-open-source-toolkit-for","title":"robo-gym -- An Open Source Toolkit for Distributed Deep Reinforcement Learning on Real and Simulated Robots","date":"2020-07-06","arxiv_id":"2007.02753","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-paraphrasing-via-deep","title":"Unsupervised Paraphrasing via Deep Reinforcement Learning","date":"2020-07-05","arxiv_id":"2007.02244","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-autonomous-free-airspace-en-route","title":"An Autonomous Free Airspace En-route Controller using Deep Reinforcement Learning Techniques","date":"2020-07-03","arxiv_id":"2007.01599","repositories_listed":0,"syntology":null},{"url":null,"slug":"dueling-deep-q-network-for-unsupervised-inter","title":"Dueling Deep Q-Network for Unsupervised Inter-frame Eye Movement Correction in Optical Coherence Tomography Volumes","date":"2020-07-03","arxiv_id":"2007.01522","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-deep-reinforcement-learning-for-1","title":"Decentralized Deep Reinforcement Learning for Network Level Traffic Signal Control","date":"2020-07-02","arxiv_id":"2007.03433","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-driven-inspection","title":"Deep reinforcement learning driven inspection and maintenance planning under incomplete information and constraints","date":"2020-07-02","arxiv_id":"2007.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-centered-collaborative-robots-with-deep","title":"Human-centered collaborative robots with deep reinforcement learning","date":"2020-07-02","arxiv_id":"2007.01009","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generalized-reinforcement-learning","title":"A Generalized Reinforcement Learning Algorithm for Online 3D Bin-Packing","date":"2020-07-01","arxiv_id":"2007.00463","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalization-of-hearing-aid-compression-by","title":"Personalization of Hearing Aid Compression by Human-In-Loop Deep Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.00192","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-approach-to-mimo","title":"Deep reinforcement learning approach to MIMO precoding problem: Optimality and Robustness","date":"2020-06-30","arxiv_id":"2006.16646","repositories_listed":0,"syntology":null},{"url":null,"slug":"testing-match-3-video-games-with-deep","title":"Testing match-3 video games with Deep Reinforcement Learning","date":"2020-06-30","arxiv_id":"2007.01137","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for","title":"Distributed Deep Reinforcement Learning for Intelligent Load Scheduling in Residential Smart Grids","date":"2020-06-29","arxiv_id":"2006.16100","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-finite-reward-automaton-inference-and","title":"Active Finite Reward Automaton Inference and Reinforcement Learning Using Queries and Counterexamples","date":"2020-06-28","arxiv_id":"2006.15714","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-uplink-beamforming-in-cell-free","title":"Distributed Uplink Beamforming in Cell-Free Networks Using Deep Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.15138","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-predictive-representations-in","title":"Learning predictive representations in autonomous driving to improve deep reinforcement learning","date":"2020-06-26","arxiv_id":"2006.15110","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-prediction-reaction-agents-for","title":"Perception-Prediction-Reaction Agents for Deep Reinforcement Learning","date":"2020-06-26","arxiv_id":"2006.15223","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-5","title":"Multi-Agent Deep Reinforcement Learning for HVAC Control in Commercial Buildings","date":"2020-06-25","arxiv_id":"2006.14156","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-overestimation-and-exploration-in-deep","title":"Some approaches used to overcome overestimation in Deep Reinforcement Learning algorithms","date":"2020-06-25","arxiv_id":"2006.14167","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-minimization-in-uav-aided-networks","title":"Energy Minimization in UAV-Aided Networks: Actor-Critic Learning for Constrained Scheduling Optimization","date":"2020-06-24","arxiv_id":"2006.13610","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-voltage-regulation-of-unbalanced","title":"Model-Free Voltage Regulation of Unbalanced Distribution Network Based on Surrogate Model and Deep Reinforcement Learning","date":"2020-06-24","arxiv_id":"2006.13992","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-control-for-radar","title":"Deep Reinforcement Learning Control for Radar Detection and Tracking in Congested Spectral Environments","date":"2020-06-23","arxiv_id":"2006.13173","repositories_listed":0,"syntology":null},{"url":null,"slug":"show-me-the-way-intrinsic-motivation-from","title":"Show me the Way: Intrinsic Motivation from Demonstrations","date":"2020-06-23","arxiv_id":"2006.12917","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-effect-of-multi-step-methods-on","title":"The Effect of Multi-step Methods on Overestimation in Deep Reinforcement Learning","date":"2020-06-23","arxiv_id":"2006.12692","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-deep-reinforcement-learning-based","title":"Accelerated Deep Reinforcement Learning Based Load Shedding for Emergency Voltage Control","date":"2020-06-22","arxiv_id":"2006.12667","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-combinatorial-optimization-with","title":"Constrained Combinatorial Optimization with Reinforcement Learning","date":"2020-06-22","arxiv_id":"2006.11984","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-causal-reinforcement","title":"Provably Efficient Causal Reinforcement Learning with Confounded Observational Data","date":"2020-06-22","arxiv_id":"2006.12311","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-cooperation-through-mutual","title":"Emergent cooperation through mutual information maximization","date":"2020-06-21","arxiv_id":"2006.11769","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for","title":"A Reinforcement Learning Approach for Transient Control of Liquid Rocket Engines","date":"2020-06-19","arxiv_id":"2006.11108","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-earn-enabling-coordination-within-a","title":"Learn to Earn: Enabling Coordination within a Ride Hailing Fleet","date":"2020-06-19","arxiv_id":"2006.10904","repositories_listed":0,"syntology":null},{"url":null,"slug":"nrowan-dqn-a-stable-noisy-network-with-noise","title":"NROWAN-DQN: A Stable Noisy Network with Noise Reduction and Online Weight Adjustment for Exploration","date":"2020-06-19","arxiv_id":"2006.10980","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-amidst-lifelong","title":"Deep Reinforcement Learning amidst Lifelong Non-Stationarity","date":"2020-06-18","arxiv_id":"2006.10701","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-estimation-bias-via-weighted-delayed","title":"WD3: Taming the Estimation Bias in Deep Reinforcement Learning","date":"2020-06-18","arxiv_id":"2006.12622","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-controller-for-3d","title":"Deep Reinforcement Learning Controller for 3D Path-following and Collision Avoidance by Autonomous Underwater Vehicles","date":"2020-06-17","arxiv_id":"2006.09792","repositories_listed":0,"syntology":null},{"url":null,"slug":"colreg-compliant-collision-avoidance-for","title":"COLREG-Compliant Collision Avoidance for Unmanned Surface Vehicle using Deep Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.09540","repositories_listed":0,"syntology":null},{"url":null,"slug":"index-selection-for-nosql-database-with-deep","title":"Index Selection for NoSQL Database with Deep Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.08842","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-order-batching-and-sequencing","title":"Solving the Order Batching and Sequencing Problem using Deep Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.09507","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-high-fidelity-multi-qubit-gates-for","title":"Designing high-fidelity multi-qubit gates for semiconductor quantum dots through deep reinforcement learning","date":"2020-06-15","arxiv_id":"2006.08813","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brief-look-at-generalization-in-visual-meta","title":"A Brief Look at Generalization in Visual Meta-Reinforcement Learning","date":"2020-06-12","arxiv_id":"2006.07262","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-for-searching-and-planning","title":"Continuous Control for Searching and Planning with a Learned Model","date":"2020-06-12","arxiv_id":"2006.07430","repositories_listed":0,"syntology":null},{"url":null,"slug":"decorrelated-double-q-learning","title":"Decorrelated Double Q-learning","date":"2020-06-12","arxiv_id":"2006.06956","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-neural","title":"Deep Reinforcement Learning for Neural Control","date":"2020-06-12","arxiv_id":"2006.07352","repositories_listed":0,"syntology":null},{"url":null,"slug":"starcraft-ii-build-order-optimization-using","title":"StarCraft II Build Order Optimization using Deep Reinforcement Learning and Monte-Carlo Tree Search","date":"2020-06-12","arxiv_id":"2006.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"systematic-generalisation-through-task","title":"Systematic Generalisation through Task Temporal Logic and Deep Reinforcement Learning","date":"2020-06-12","arxiv_id":"2006.08767","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-electric","title":"Deep Reinforcement Learning for Electric Transmission Voltage Control","date":"2020-06-11","arxiv_id":"2006.06728","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-optical","title":"Deep reinforcement learning for optical systems: A case study of mode-locked lasers","date":"2020-06-10","arxiv_id":"2006.05579","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-cost-management-in-smart-meters-an","title":"Privacy-Cost Management in Smart Meters with Mutual Information-Based Reinforcement Learning","date":"2020-06-10","arxiv_id":"2006.06106","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-non-stationarity-on","title":"Transient Non-Stationarity and Generalisation in Deep Reinforcement Learning","date":"2020-06-10","arxiv_id":"2006.05826","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealing-deep-reinforcement-learning-models","title":"Stealing Deep Reinforcement Learning Models for Fun and Profit","date":"2020-06-09","arxiv_id":"2006.05032","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-temporal-difference-and-q-learning-learn","title":"Can Temporal-Difference and Q-Learning Learn Representation? A Mean-Field Theory","date":"2020-06-08","arxiv_id":"2006.04761","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-data-poisoning-attacks","title":"Online Data Poisoning Attacks","date":"2020-06-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"randomized-policy-learning-for-continuous","title":"Randomized Policy Learning for Continuous State and Action MDPs","date":"2020-06-08","arxiv_id":"2006.04331","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoprivacy-automated-layer-wise-parameter","title":"AutoPrivacy: Automated Layer-wise Parameter Selection for Secure Neural Network Inference","date":"2020-06-07","arxiv_id":"2006.04219","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-poverty-mapping-using-deep","title":"Efficient Poverty Mapping using Deep Reinforcement Learning","date":"2020-06-07","arxiv_id":"2006.04224","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-model-calibration-with-deep","title":"Real-Time Model Calibration with Deep Reinforcement Learning","date":"2020-06-07","arxiv_id":"2006.04001","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-deep-reinforcement-learning-3","title":"A Multi-Agent Deep Reinforcement Learning Method for Cooperative Load Frequency Control of a Multi-Area Power System","date":"2020-06-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-scan-a-deep-reinforcement","title":"Learning to Scan: A Deep Reinforcement Learning Approach for Personalized Scanning in CT Imaging","date":"2020-06-03","arxiv_id":"2006.02420","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-on-reinforcement-learning","title":"Adversarial Attacks on Reinforcement Learning based Energy Management Systems of Extended Range Electric Delivery Vehicles","date":"2020-06-01","arxiv_id":"2006.00817","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-multi-robot-collision-avoidance","title":"Distributed multi-robot collision avoidance via deep reinforcement learning for navigation in complex scenarios","date":"2020-05-31","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-voltage-regulation-of-active","title":"Distributed Voltage Regulation of Active Distribution System Based on Enhanced Multi-agent Deep Reinforcement Learning","date":"2020-05-31","arxiv_id":"2006.00546","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-knowledge-integration-by-gradient","title":"Domain Knowledge Integration By Gradient Matching For Sample-Efficient Reinforcement Learning","date":"2020-05-28","arxiv_id":"2005.13778","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-residential-energy-management","title":"Intelligent Residential Energy Management System using Deep Reinforcement Learning","date":"2020-05-28","arxiv_id":"2005.14259","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-adversarial-resilience-learning","title":"The Adversarial Resilience Learning Architecture for AI-based Modelling, Exploration, and Operation of Complex Cyber-Physical Systems","date":"2020-05-27","arxiv_id":"2005.13601","repositories_listed":0,"syntology":null},{"url":null,"slug":"anomaly-detection-under-controlled-sensing","title":"Anomaly Detection Under Controlled Sensing Using Actor-Critic Reinforcement Learning","date":"2020-05-26","arxiv_id":"2006.01044","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-leo-satellite-and-uav-relaying","title":"Integrating LEO Satellite and UAV Relaying via Reinforcement Learning for Non-Terrestrial Networks","date":"2020-05-26","arxiv_id":"2005.12521","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-intervention-centric-causal-reasoning","title":"Towards intervention-centric causal reasoning in learning agents","date":"2020-05-26","arxiv_id":"2005.12968","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-power-1","title":"Deep Reinforcement Learning Based Power Allocation for D2D Network","date":"2020-05-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-value-estimation-for-single-task","title":"Dynamic Value Estimation for Single-Task Multi-Scene Reinforcement Learning","date":"2020-05-25","arxiv_id":"2005.12254","repositories_listed":0,"syntology":null},{"url":null,"slug":"formal-methods-with-a-touch-of-magic","title":"Formal Methods with a Touch of Magic","date":"2020-05-25","arxiv_id":"2005.12175","repositories_listed":0,"syntology":null},{"url":null,"slug":"generator-and-critic-a-deep-reinforcement","title":"Generator and Critic: A Deep Reinforcement Learning Approach for Slate Re-ranking in E-commerce","date":"2020-05-25","arxiv_id":"2005.12206","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-monitored-reinforcement-learning","title":"Gradient Monitored Reinforcement Learning","date":"2020-05-25","arxiv_id":"2005.12108","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-driven-deep-reinforcement","title":"Optimization-driven Deep Reinforcement Learning for Robust Beamforming in IRS-assisted Wireless Communications","date":"2020-05-25","arxiv_id":"2005.11885","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-entropy-for-out-of-distribution","title":"Policy Entropy for Out-of-Distribution Classification","date":"2020-05-25","arxiv_id":"2005.12069","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-agent-optimization-through-policy","title":"Single-Agent Optimization Through Policy Iteration Using Monte-Carlo Tree Search","date":"2020-05-22","arxiv_id":"2005.11335","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-resource-scheduling-for-large","title":"Distributed Resource Scheduling for Large-Scale MEC Systems: A Multi-Agent Ensemble Deep Reinforcement Learning with Imitation Acceleration","date":"2020-05-21","arxiv_id":"2005.12364","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-high-level","title":"Deep Reinforcement Learning for High Level Character Control","date":"2020-05-20","arxiv_id":"2005.10391","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-deep-reinforcement-learning-for","title":"Two-stage Deep Reinforcement Learning for Inverter-based Volt-VAR Control in Active Distribution Networks","date":"2020-05-20","arxiv_id":"2005.11142","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-instruction-following-with-deep","title":"Human Instruction-Following with Deep Reinforcement Learning via Transfer-Learning from Text","date":"2020-05-19","arxiv_id":"2005.09382","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-herd-agents-amongst-obstacles","title":"Learning to Herd Agents Amongst Obstacles: Training Robust Shepherding Behaviors using Deep Reinforcement Learning","date":"2020-05-19","arxiv_id":"2005.09476","repositories_listed":0,"syntology":null},{"url":null,"slug":"prototypical-q-networks-for-automatic","title":"Prototypical Q Networks for Automatic Conversational Diagnosis and Few-Shot New Disease Adaption","date":"2020-05-19","arxiv_id":"2005.11153","repositories_listed":0,"syntology":null},{"url":null,"slug":"basal-glucose-control-in-type-1-diabetes","title":"Basal Glucose Control in Type 1 Diabetes using Deep Reinforcement Learning: An In Silico Validation","date":"2020-05-18","arxiv_id":"2005.09059","repositories_listed":0,"syntology":null},{"url":null,"slug":"dampen-the-stop-and-go-traffic-with-connected","title":"Dampen the Stop-and-Go Traffic with Connected and Automated Vehicles -- A Deep Reinforcement Learning Approach","date":"2020-05-17","arxiv_id":"2005.08245","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-prediction-rendering-and","title":"Learning-based Prediction, Rendering and Association Optimization for MEC-enabled Wireless Virtual Reality (VR) Network","date":"2020-05-17","arxiv_id":"2005.08332","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-learning-in-deep-reinforcement","title":"Learning Transferable Concepts in Deep Reinforcement Learning","date":"2020-05-16","arxiv_id":"2005.07870","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-multiagent-control-using","title":"Continuous Multiagent Control using Collective Behavior Entropy for Large-Scale Home Energy Management","date":"2020-05-14","arxiv_id":"2005.10000","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-guarantees-for-safe-deep","title":"Probabilistic Guarantees for Safe Deep Reinforcement Learning","date":"2020-05-14","arxiv_id":"2005.07073","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-coloring-for-end-to-end-instance","title":"Reinforced Coloring for End-to-End Instance Segmentation","date":"2020-05-14","arxiv_id":"2005.07058","repositories_listed":0,"syntology":null},{"url":null,"slug":"solve-traveling-salesman-problem-by-monte","title":"Solve Traveling Salesman Problem by Monte Carlo Tree Search and Deep Neural Network","date":"2020-05-14","arxiv_id":"2005.06879","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealthy-and-efficient-adversarial-attacks","title":"Stealthy and Efficient Adversarial Attacks against Deep Reinforcement Learning","date":"2020-05-14","arxiv_id":"2005.07099","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-simulation-to-real-world-maneuver","title":"From Simulation to Real World Maneuver Execution using Deep Reinforcement Learning","date":"2020-05-13","arxiv_id":"2005.07023","repositories_listed":0,"syntology":null},{"url":null,"slug":"proxy-experience-replay-federated","title":"Proxy Experience Replay: Federated Distillation for Distributed Reinforcement Learning","date":"2020-05-13","arxiv_id":"2005.06105","repositories_listed":0,"syntology":null},{"url":null,"slug":"unbiased-deep-reinforcement-learning-a","title":"Unbiased Deep Reinforcement Learning: A General Training Framework for Existing and Future Algorithms","date":"2020-05-12","arxiv_id":"2005.07782","repositories_listed":0,"syntology":null}],"record_sha256":"5c8d63966cc38a90dd033f2bc4b2d357a28196ca812a8d7357c77f7ce8b49e49","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}