{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/46","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":46,"pages_in_order":59,"rows_per_page":100,"rows":[4501,4600],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/45","next":"/task/deep-reinforcement-learning/papers/47","papers":[{"url":null,"slug":"deep-reinforcement-learning-based-dynamic-1","title":"Deep Reinforcement Learning Based Dynamic Route Planning for Minimizing Travel Time","date":"2020-11-03","arxiv_id":"2011.01771","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-heterogeneous-deep-reinforcement","title":"Cooperative Heterogeneous Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.00791","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-electricity","title":"Deep Reinforcement Learning in Electricity Generation Investment for the Minimization of Long-Term Carbon Emissions and Electricity Costs","date":"2020-11-02","arxiv_id":"2011.02342","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-multiple-intelligent-reflecting","title":"Multi-IRS-assisted Multi-Cell Uplink MIMO Communications under Imperfect CSI: A Deep Reinforcement Learning Approach","date":"2020-11-02","arxiv_id":"2011.01141","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-deep-reinforcement-learning-policy","title":"Learning a Deep Reinforcement Learning Policy Over the Latent Space of a Pre-trained GAN for Semantic Age Manipulation","date":"2020-11-02","arxiv_id":"2011.00954","repositories_listed":0,"syntology":null},{"url":null,"slug":"observation-space-matters-benchmark-and","title":"Observation Space Matters: Benchmark and Optimization Algorithm","date":"2020-11-02","arxiv_id":"2011.00756","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-double-critic-incorporating-model-based","title":"Actor-Double-Critic: Incorporating Model-Based Critic for Task-Oriented Dialogue Systems","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bound-controller-for-a-quadruped-robot-using","title":"Efficient Learning of Control Policies for Robust Quadruped Bounding using Pretrained Neural Networks","date":"2020-11-01","arxiv_id":"2011.00446","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-a-robot-trust-you-a-drl-based-approach-to","title":"Can a Robot Trust You? A DRL-Based Approach to Trust-Driven Human-Guided Navigation","date":"2020-11-01","arxiv_id":"2011.00554","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-when-to-switch-composing-controllers","title":"Learning When to Switch: Composing Controllers to Traverse a Sequence of Terrain Artifacts","date":"2020-11-01","arxiv_id":"2011.00440","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reactive-planning-in-dynamic","title":"Deep Reactive Planning in Dynamic Environments","date":"2020-10-31","arxiv_id":"2011.00155","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-preserving-synthetic-news-generation-an","title":"Topic-Preserving Synthetic News Generation: An Adversarial Deep Reinforcement Learning Approach","date":"2020-10-30","arxiv_id":"2010.16324","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-versus-machine-attention-in-deep","title":"Machine versus Human Attention in Deep Reinforcement Learning Tasks","date":"2020-10-29","arxiv_id":"2010.15942","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepfoldit-a-deep-reinforcement-learning","title":"DeepFoldit -- A Deep Reinforcement Learning Neural Network Folding Proteins","date":"2020-10-28","arxiv_id":"2011.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-interpretable-approximations-to","title":"Designing Interpretable Approximations to Deep Reinforcement Learning","date":"2020-10-28","arxiv_id":"2010.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-reinforcement-learning-for-continuous","title":"Can Reinforcement Learning for Continuous Control Generalize Across Physics Engines?","date":"2020-10-27","arxiv_id":"2010.14444","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-decision-making-for-urban","title":"Behavioral decision-making for urban autonomous driving in the presence of pedestrians using Deep Recurrent Q-Network","date":"2020-10-26","arxiv_id":"2010.13407","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-state","title":"Lyapunov-Based Reinforcement Learning State Estimator","date":"2020-10-26","arxiv_id":"2010.13529","repositories_listed":0,"syntology":null},{"url":null,"slug":"pairwise-heuristic-sequence-alignment","title":"Pairwise heuristic sequence alignment algorithm based on deep reinforcement learning","date":"2020-10-26","arxiv_id":"2010.13478","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-federated-learning-and-digital-twin","title":"Adaptive Federated Learning and Digital Twin for Industrial Internet of Things","date":"2020-10-25","arxiv_id":"2010.13058","repositories_listed":0,"syntology":null},{"url":null,"slug":"xlvin-executed-latent-value-iteration-nets-1","title":"XLVIN: eXecuted Latent Value Iteration Nets","date":"2020-10-25","arxiv_id":"2010.13146","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-exploration-of-deep","title":"Improving the Exploration of Deep Reinforcement Learning in Continuous Domains using Planning for Policy Search","date":"2020-10-24","arxiv_id":"2010.12974","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-on-deep-algorithmic","title":"Adversarial Attacks on Deep Algorithmic Trading Policies","date":"2020-10-22","arxiv_id":"2010.11388","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-planner-augmented-reinforcement","title":"Motion Planner Augmented Reinforcement Learning for Robot Manipulation in Obstructed Environments","date":"2020-10-22","arxiv_id":"2010.11940","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-surrogate-q-learning-for-autonomous","title":"Deep Surrogate Q-Learning for Autonomous Driving","date":"2020-10-21","arxiv_id":"2010.11278","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-graph-optimizers-for-ml","title":"Transferable Graph Optimizers for ML Compilers","date":"2020-10-21","arxiv_id":"2010.12438","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-lane-merge","title":"Deep Reinforcement Learning in Lane Merge Coordination for Connected Vehicles","date":"2020-10-20","arxiv_id":"2010.10567","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-leo-satellites-and-multi-uav","title":"Integrating LEO Satellites and Multi-UAV Reinforcement Learning for Hybrid FSO/RF Non-Terrestrial Networks","date":"2020-10-20","arxiv_id":"2010.10138","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-navigate-maplessly-with-varied-lidar","title":"Learn to Navigate Maplessly with Varied LiDAR Configurations: A Support Point-Based Approach","date":"2020-10-20","arxiv_id":"2010.10209","repositories_listed":0,"syntology":null},{"url":null,"slug":"negotiating-team-formation-using-deep-1","title":"Negotiating Team Formation Using Deep Reinforcement Learning","date":"2020-10-20","arxiv_id":"2010.10380","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-of-service-based-radar-resource","title":"Quality of service based radar resource management using deep reinforcement learning","date":"2020-10-20","arxiv_id":"2010.10210","repositories_listed":0,"syntology":null},{"url":null,"slug":"chance-constrained-control-with-lexicographic","title":"Chance-Constrained Control with Lexicographic Deep Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09468","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-adaptive-2","title":"Deep Reinforcement Learning for Adaptive Network Slicing in 5G for Intelligent Vehicular Systems and Smart Cities","date":"2020-10-19","arxiv_id":"2010.09916","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-safety-of-deep-reinforcement","title":"Evaluating the Safety of Deep Reinforcement Learning Models using Semi-Formal Verification","date":"2020-10-19","arxiv_id":"2010.09387","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-algorithms-for-graph-navigation","title":"Neural Algorithms for Graph Navigation","date":"2020-10-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-large-neighborhood-search","title":"Neural Large Neighborhood Search","date":"2020-10-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-evolution-strategies-pipeline-for","title":"Scalable Evolution Strategies Pipeline for Solving the Vehicle Routing Problem","date":"2020-10-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-dynamic-for-self-supervised","title":"Variational Dynamic for Self-Supervised Exploration in Deep Reinforcement Learning","date":"2020-10-17","arxiv_id":"2010.08755","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-control-of-a-particle-accelerator","title":"Autonomous Control of a Particle Accelerator using Deep Reinforcement Learning","date":"2020-10-16","arxiv_id":"2010.08141","repositories_listed":0,"syntology":null},{"url":null,"slug":"doom-a-novel-adversarial-drl-based-op-code","title":"DOOM: A Novel Adversarial-DRL-Based Op-Code Level Metamorphic Malware Obfuscator for the Enhancement of IDS","date":"2020-10-16","arxiv_id":"2010.08608","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-robotic-object-search-via-hiem","title":"Efficient Robotic Object Search via HIEM: Hierarchical Policy Learning with Intrinsic-Extrinsic Modeling","date":"2020-10-16","arxiv_id":"2010.08596","repositories_listed":0,"syntology":null},{"url":null,"slug":"revenue-and-energy-efficiency-driven-delay","title":"Revenue and Energy Efficiency-Driven Delay Constrained Computing Task Offloading and Resource Allocation in a Vehicular Edge Computing Network: A Deep Reinforcement Learning Approach","date":"2020-10-16","arxiv_id":"2010.08119","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-nesterov-s-accelerated-quasi-newton-method","title":"A Nesterov's Accelerated quasi-Newton method for Global Routing using Deep Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.09465","repositories_listed":0,"syntology":null},{"url":null,"slug":"applicability-and-challenges-of-deep","title":"Applicability and Challenges of Deep Reinforcement Learning for Satellite Frequency Plan Design","date":"2020-10-15","arxiv_id":"2010.08015","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-control-adaptation-with","title":"Energy-Efficient Control Adaptation with Safety Guarantees for Learning-Enabled Cyber-Physical Systems","date":"2020-10-15","arxiv_id":"2008.06162","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and","title":"Deep Reinforcement Learning and Transportation Research: A Comprehensive Review","date":"2020-10-13","arxiv_id":"2010.06187","repositories_listed":0,"syntology":null},{"url":null,"slug":"grid-interactive-multi-zone-building-control","title":"Grid-Interactive Multi-Zone Building Control Using Reinforcement Learning with Global-Local Policy Search","date":"2020-10-13","arxiv_id":"2010.06718","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-echo-state-q-network-deqn-and-its","title":"Deep Echo State Q-Network (DEQN) and Its Application in Dynamic Spectrum Sharing for 5G and Beyond","date":"2020-10-12","arxiv_id":"2010.05449","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-scheduling","title":"Deep-Reinforcement-Learning-Based Scheduling with Contiguous Resource Allocation for Next-Generation Cellular Systems","date":"2020-10-11","arxiv_id":"2010.11269","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-asset","title":"Deep Reinforcement Learning for Asset Allocation in US Equities","date":"2020-10-09","arxiv_id":"2010.04404","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-locomote-understanding-how","title":"Learning to Locomote: Understanding How Environment Design Matters for Deep Reinforcement Learning","date":"2020-10-09","arxiv_id":"2010.04304","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-curriculum-learning-for-walking-over","title":"Guided Curriculum Learning for Walking Over Complex Terrain","date":"2020-10-08","arxiv_id":"2010.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-intrinsic-symbolic-rewards-in-1","title":"Learning Intrinsic Symbolic Rewards in Reinforcement Learning","date":"2020-10-08","arxiv_id":"2010.03694","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-algorithm-for-high-dimensional","title":"Actor-Critic Algorithm for High-dimensional Partial Differential Equations","date":"2020-10-07","arxiv_id":"2010.03647","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-safety-assurance-for-deep","title":"Online Safety Assurance for Deep Reinforcement Learning","date":"2020-10-07","arxiv_id":"2010.03625","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-optimization-with-relative","title":"Proximal Policy Optimization with Relative Pearson Divergence","date":"2020-10-07","arxiv_id":"2010.03290","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-dynamic-3","title":"Deep Reinforcement Learning-Based Dynamic Resource Management for Mobile Edge Computing in Industrial Internet of Things","date":"2020-10-06","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributed-model-free-ride-sharing","title":"A Distributed Model-Free Ride-Sharing Approach for Joint Matching, Pricing, and Dispatching using Deep Reinforcement Learning","date":"2020-10-05","arxiv_id":"2010.01755","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-actor-dual-critic-model-for-remote","title":"A Novel Actor Dual-Critic Model for Remote Sensing Image Captioning","date":"2020-10-05","arxiv_id":"2010.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-collaborative","title":"Deep Reinforcement Learning for Collaborative Edge Computing in Vehicular Networks","date":"2020-10-05","arxiv_id":"2010.01722","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-electric-1","title":"Deep Reinforcement Learning for Electric Vehicle Routing Problem with Time Windows","date":"2020-10-05","arxiv_id":"2010.02068","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-delay","title":"Deep Reinforcement Learning for Delay-Oriented IoT Task Scheduling in Space-Air-Ground Integrated Network","date":"2020-10-04","arxiv_id":"2010.01471","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-cost-sensitive-methods-for-identifying","title":"Test-Cost Sensitive Methods for Identifying Nearby Points","date":"2020-10-04","arxiv_id":"2010.03962","repositories_listed":0,"syntology":null},{"url":null,"slug":"attractor-selection-in-nonlinear-energy","title":"Attractor Selection in Nonlinear Energy Harvesting Using Deep Reinforcement Learning","date":"2020-10-03","arxiv_id":"2010.01255","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-tabula-rasa-a-modular-reinforcement","title":"Beyond Tabula-Rasa: a Modular Reinforcement Learning Approach for Physically Embedded 3D Sokoban","date":"2020-10-03","arxiv_id":"2010.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"pomdps-in-continuous-time-and-discrete-spaces","title":"POMDPs in Continuous Time and Discrete Spaces","date":"2020-10-02","arxiv_id":"2010.01014","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-mixed","title":"Deep Reinforcement Learning with Mixed Convolutional Network","date":"2020-10-01","arxiv_id":"2010.00717","repositories_listed":0,"syntology":null},{"url":null,"slug":"aamdrl-augmented-asset-management-with-deep","title":"AAMDRL: Augmented Asset Management with Deep Reinforcement Learning","date":"2020-09-30","arxiv_id":"2010.08497","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-markowitz-planning","title":"Bridging the gap between Markowitz planning and deep reinforcement learning","date":"2020-09-30","arxiv_id":"2010.09108","repositories_listed":0,"syntology":null},{"url":null,"slug":"facilitating-connected-autonomous-vehicle","title":"Facilitating Connected Autonomous Vehicle Operations Using Space-weighted Information Fusion and Deep Reinforcement Learning Based Control","date":"2020-09-30","arxiv_id":"2009.14665","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategy-and-benchmark-for-converting-deep-q","title":"Strategy and Benchmark for Converting Deep Q-Networks to Event-Driven Spiking Neural Networks","date":"2020-09-30","arxiv_id":"2009.14456","repositories_listed":0,"syntology":null},{"url":null,"slug":"toolpath-design-for-additive-manufacturing","title":"Toolpath design for additive manufacturing using deep reinforcement learning","date":"2020-09-30","arxiv_id":"2009.14365","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-design-space-adaptation-with-deep","title":"Trust-Region Method with Deep Reinforcement Learning in Analog Design Space Exploration","date":"2020-09-29","arxiv_id":"2009.13772","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-discretization-for-continuous","title":"Adaptive Discretization for Continuous Control using Particle Filtering Policy Network","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-der-cyber","title":"Deep Reinforcement Learning for DER Cyber-Attack Mitigation","date":"2020-09-28","arxiv_id":"2009.13088","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-information-opponent-modelling-using","title":"Local Information Opponent Modelling Using Variational Autoencoders","date":"2020-09-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-in-event-triggered-control","title":"Machine Learning in Event-Triggered Control: Recent Advances and Open Issues","date":"2020-09-27","arxiv_id":"2009.12783","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-deep-reinforcement-learning-for-ride","title":"Scalable Deep Reinforcement Learning for Ride-Hailing","date":"2020-09-27","arxiv_id":"2009.14679","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-and-power-control-for-wireless","title":"Scheduling and Power Control for Wireless Multicast Systems via Deep Reinforcement Learning","date":"2020-09-27","arxiv_id":"2011.14799","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-neural-induction-of-value-iteration","title":"Graph neural induction of value iteration","date":"2020-09-26","arxiv_id":"2009.12604","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-rational-control-with-partially-1","title":"Inverse Rational Control with Partially Observable Continuous Nonlinear Dynamics","date":"2020-09-26","arxiv_id":"2009.12576","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-stage","title":"Deep Reinforcement Learning with a Stage Incentive Mechanism of Dense Reward for Robotic Trajectory Planning","date":"2020-09-25","arxiv_id":"2009.12068","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-in-deep-reinforcement","title":"Sim-to-Real Transfer in Deep Reinforcement Learning for Robotics: a Survey","date":"2020-09-24","arxiv_id":"2009.13303","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-deep-reinforcement-learning","title":"A Multi-Agent Deep Reinforcement Learning Approach for a Distributed Energy Marketplace in Smart Grids","date":"2020-09-23","arxiv_id":"2009.10905","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-based","title":"Multi-Agent Deep Reinforcement Learning Based Trajectory Planning for Multi-UAV Assisted Mobile Edge Computing","date":"2020-09-23","arxiv_id":"2009.11277","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-based","title":"Robust Reinforcement Learning-based Autonomous Driving Agent for Simulation and Real World","date":"2020-09-23","arxiv_id":"2009.11212","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-on-line","title":"Deep Reinforcement Learning for On-line Dialogue State Tracking","date":"2020-09-22","arxiv_id":"2009.10321","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-structured-actor-critic","title":"Distributed Structured Actor-Critic Reinforcement Learning for Universal Dialogue Management","date":"2020-09-22","arxiv_id":"2009.10326","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-hierarchical-dialogue-policy-with","title":"Structured Hierarchical Dialogue Policy with Graph Neural Networks","date":"2020-09-22","arxiv_id":"2009.10355","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-deep-reinforcement-learning-meets","title":"When Deep Reinforcement Learning Meets Federated Learning: Intelligent Multi-Timescale Resource Management for Multi-access Edge Computing in 5G Ultra Dense Network","date":"2020-09-22","arxiv_id":"2009.10601","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-beamforming-for-ris-empowered-multi","title":"Hybrid Beamforming for RIS-Empowered Multi-hop Terahertz Communications: A DRL-based Method","date":"2020-09-20","arxiv_id":"2009.09380","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-ai-policies-using-evolutionary","title":"Towards Interpretable-AI Policies Induction using Evolutionary Nonlinear Decision Trees for Discrete Action Systems","date":"2020-09-20","arxiv_id":"2009.09521","repositories_listed":0,"syntology":null},{"url":null,"slug":"lyapunov-based-reinforcement-learning-for","title":"Lyapunov-Based Reinforcement Learning for Decentralized Multi-Agent Control","date":"2020-09-20","arxiv_id":"2009.09361","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-closed-loop","title":"Deep Reinforcement Learning for Closed-Loop Blood Glucose Control","date":"2020-09-18","arxiv_id":"2009.09051","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-assisted-deep-reinforcement","title":"Knowledge-Assisted Deep Reinforcement Learning in 5G Scheduler Design: From Theoretical Framework to Implementation","date":"2020-09-17","arxiv_id":"2009.08346","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnable-strategies-for-bilateral-agent","title":"Learnable Strategies for Bilateral Agent Negotiation over Multiple Issues","date":"2020-09-17","arxiv_id":"2009.08302","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-behavior-level-explanation-for-deep","title":"Reconstructing Actions To Explain Deep Reinforcement Learning","date":"2020-09-17","arxiv_id":"2009.08507","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-fas-a-novel-framework-based-on-deep","title":"DRL-FAS: A Novel Framework Based on Deep Reinforcement Learning for Face Anti-Spoofing","date":"2020-09-16","arxiv_id":"2009.07529","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-your-hedge-with-deep-reinforcement","title":"Time your hedge with Deep Reinforcement Learning","date":"2020-09-16","arxiv_id":"2009.14136","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-in-deep-reinforcement","title":"Transfer Learning in Deep Reinforcement Learning: A Survey","date":"2020-09-16","arxiv_id":"2009.07888","repositories_listed":0,"syntology":null}],"record_sha256":"009c2d89b70514cb6d2d8dc0400b49581e8644a93e2d918566e16b8a67a823ae","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}