{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/54","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":54,"pages_in_order":59,"rows_per_page":100,"rows":[5301,5400],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/53","next":"/task/deep-reinforcement-learning/papers/55","papers":[{"url":null,"slug":"rating-continuous-actions-in-spatial-multi","title":"Rating Continuous Actions in Spatial Multi-Agent Problems","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-consistent-performance-on-atari-using","title":"Towards Consistent Performance on Atari using Expert Demonstrations","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-q-learning-method-for-downlink-power","title":"A Deep Q-Learning Method for Downlink Power Allocation in Multi-Cell Networks","date":"2019-04-30","arxiv_id":"1904.13032","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-imagination-for-sample","title":"Generative Adversarial Imagination for Sample Efficient Deep Reinforcement Learning","date":"2019-04-30","arxiv_id":"1904.13255","repositories_listed":0,"syntology":null},{"url":null,"slug":"arbitrage-of-energy-storage-in-electricity","title":"Arbitrage of Energy Storage in Electricity Markets with Deep Reinforcement Learning","date":"2019-04-28","arxiv_id":"1904.12232","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-optimal-1","title":"Deep Reinforcement Learning for Optimal Critical Care Pain Management with Morphine using Dueling Double-Deep Q Networks","date":"2019-04-25","arxiv_id":"1904.11115","repositories_listed":0,"syntology":null},{"url":null,"slug":"ray-interference-a-source-of-plateaus-in-deep","title":"Ray Interference: a Source of Plateaus in Deep Reinforcement Learning","date":"2019-04-25","arxiv_id":"1904.11455","repositories_listed":0,"syntology":null},{"url":null,"slug":"190602671","title":"Grounding Natural Language Commands to StarCraft II Game States for Narration-Guided Reinforcement Learning","date":"2019-04-24","arxiv_id":"1906.02671","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-voltage-control-for-grid-operation","title":"Autonomous Voltage Control for Grid Operation Using Deep Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.10597","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-you-act-tells-a-lot-privacy-leakage","title":"How You Act Tells a Lot: Privacy-Leakage Attack on Deep Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.11082","repositories_listed":0,"syntology":null},{"url":null,"slug":"driving-decision-and-control-for-autonomous","title":"Driving Decision and Control for Autonomous Lane Change based on Deep Reinforcement Learning","date":"2019-04-23","arxiv_id":"1904.10171","repositories_listed":0,"syntology":null},{"url":null,"slug":"190501357","title":"Teaching on a Budget in Multi-Agent Deep Reinforcement Learning","date":"2019-04-19","arxiv_id":"1905.01357","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-autonomous-braking-system","title":"Multi-Objective Autonomous Braking System using Naturalistic Dataset","date":"2019-04-15","arxiv_id":"1904.07705","repositories_listed":0,"syntology":null},{"url":null,"slug":"saliency-prediction-on-omnidirectional-images","title":"Saliency Prediction on Omnidirectional Images with Generative Adversarial Imitation Learning","date":"2019-04-15","arxiv_id":"1904.07080","repositories_listed":0,"syntology":null},{"url":null,"slug":"dot-to-dot-achieving-structured-robotic","title":"Dot-to-Dot: Explainable Hierarchical Reinforcement Learning for Robotic Manipulation","date":"2019-04-14","arxiv_id":"1904.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-scheduling-function-design-in-sdn","title":"Effective Scheduling Function Design in SDN through Deep Reinforcement Learning","date":"2019-04-12","arxiv_id":"1904.06039","repositories_listed":0,"syntology":null},{"url":null,"slug":"safer-deep-rl-with-shallow-mcts-a-case-study","title":"Safer Deep RL with Shallow MCTS: A Case Study in Pommerman","date":"2019-04-10","arxiv_id":"1904.05759","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-pro-level-ai-for-real-time-fighting","title":"Creating Pro-Level AI for a Real-Time Fighting Game Using Deep Reinforcement Learning","date":"2019-04-08","arxiv_id":"1904.03821","repositories_listed":0,"syntology":null},{"url":null,"slug":"jam-me-if-you-can-defeating-jammer-with-deep","title":"\"Jam Me If You Can'': Defeating Jammer with Deep Dueling Neural Network Architecture and Ambient Backscattering Augmented Communications","date":"2019-04-08","arxiv_id":"1904.03897","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-agents-for-physical-construction","title":"Structured agents for physical construction","date":"2019-04-05","arxiv_id":"1904.03177","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-a-robot-become-a-movie-director-learning","title":"Can a Robot Become a Movie Director? Learning Artistic Principles for Aerial Cinematography","date":"2019-04-04","arxiv_id":"1904.02579","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-and-visualizing-weaknesses-of-deep","title":"Finding and Visualizing Weaknesses of Deep Reinforcement Learning Agents","date":"2019-04-02","arxiv_id":"1904.01318","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-cancer-chemotherapy-schedule-a","title":"Personalized Cancer Chemotherapy Schedule: a numerical comparison of performance and robustness in model-based and model-free scheduling methodologies","date":"2019-04-02","arxiv_id":"1904.01200","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-power-control-for-large-energy","title":"Distributed Power Control for Large Energy Harvesting Networks: A Multi-Agent Deep Reinforcement Learning Approach","date":"2019-04-01","arxiv_id":"1904.00601","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-multi-agent-reinforcement-1","title":"Cooperative Multi-Agent Reinforcement Learning Framework for Scalping Trading","date":"2019-03-31","arxiv_id":"1904.00441","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-averse-robust-adversarial-reinforcement","title":"Risk Averse Robust Adversarial Reinforcement Learning","date":"2019-03-31","arxiv_id":"1904.00511","repositories_listed":0,"syntology":null},{"url":null,"slug":"lane-change-decision-making-through-deep","title":"Lane Change Decision-making through Deep Reinforcement Learning with Rule-based Constraints","date":"2019-03-30","arxiv_id":"1904.00231","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-highway-driving-using-deep","title":"Autonomous Highway Driving using Deep Reinforcement Learning","date":"2019-03-29","arxiv_id":"1904.00035","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-control-of-stochastic-evolution-a","title":"Dynamic Control of Stochastic Evolution: A Deep Reinforcement Learning Approach to Adaptively Targeting Emergent Drug Resistance","date":"2019-03-27","arxiv_id":"1903.11373","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-where-to-see-a-novel-attention-model","title":"Learning Where to See: A Novel Attention Model for Automated Immunohistochemical Scoring","date":"2019-03-26","arxiv_id":"1903.10762","repositories_listed":0,"syntology":null},{"url":null,"slug":"winning-isnt-everything-training-human-like","title":"Winning Isn't Everything: Enhancing Game Development with Intelligent Agents","date":"2019-03-25","arxiv_id":"1903.10545","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-promoting-deep-reinforcement","title":"Diversity-Promoting Deep Reinforcement Learning for Interactive Recommendation","date":"2019-03-19","arxiv_id":"1903.07826","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with","title":"Deep Reinforcement Learning with Decorrelation","date":"2019-03-18","arxiv_id":"1903.07765","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-for-autonomous","title":"Robust Reinforcement Learning for Autonomous Driving","date":"2019-03-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"online-antenna-tuning-in-heterogeneous","title":"Online Antenna Tuning in Heterogeneous Cellular Networks with Deep Reinforcement Learning","date":"2019-03-15","arxiv_id":"1903.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-design-through-deep","title":"Task-oriented Design through Deep Reinforcement Learning","date":"2019-03-13","arxiv_id":"1903.05271","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-with-1","title":"Deep Multi-Agent Reinforcement Learning with Discrete-Continuous Hybrid Action Spaces","date":"2019-03-12","arxiv_id":"1903.04959","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-of-volume-guided","title":"Deep Reinforcement Learning of Volume-guided Progressive View Inpainting for 3D Point Scene Completion from a Single Depth Image","date":"2019-03-10","arxiv_id":"1903.04019","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeppool-distributed-model-free-algorithm-for","title":"DeepPool: Distributed Model-free Algorithm for Ride-sharing using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03882","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cooperative-game-for-automated-learning-of","title":"A cooperative game for automated learning of elasto-plasticity knowledge graphs and models with AI-guided experimentation","date":"2019-03-08","arxiv_id":"1903.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-guided-deep-reinforcement-learning-via","title":"Safety-Guided Deep Reinforcement Learning via Online Gaussian Process Estimation","date":"2019-03-06","arxiv_id":"1903.02526","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-chemical-plant-operation","title":"Synthesizing Chemical Plant Operation Procedures using Knowledge, Dynamic Simulation and Deep Reinforcement Learning","date":"2019-03-06","arxiv_id":"1903.02183","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-data-poisoning-attack","title":"Online Data Poisoning Attack","date":"2019-03-05","arxiv_id":"1903.01666","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-chinese-checkers-with","title":"Towards Understanding Chinese Checkers with Heuristics, Monte Carlo Tree Search, and Deep Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.01747","repositories_listed":0,"syntology":null},{"url":null,"slug":"microscopic-traffic-simulation-by-cooperative","title":"Microscopic Traffic Simulation by Cooperative Multi-agent Deep Reinforcement Learning","date":"2019-03-04","arxiv_id":"1903.01365","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnidrl-robust-pedestrian-detection-using","title":"OmniDRL: Robust Pedestrian Detection using Deep Reinforcement Learning on Omnidirectional Cameras","date":"2019-03-02","arxiv_id":"1903.00676","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-caching-via-deep-reinforcement","title":"Deep Reinforcement Learning for Adaptive Caching in Hierarchical Content Delivery Networks","date":"2019-02-27","arxiv_id":"1902.10301","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-packet-classification","title":"Neural Packet Classification","date":"2019-02-27","arxiv_id":"1902.10319","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-meta-interpretive-learning-outperform","title":"Can Meta-Interpretive Learning outperform Deep Reinforcement Learning of Evaluable Game strategies?","date":"2019-02-26","arxiv_id":"1902.09835","repositories_listed":0,"syntology":null},{"url":null,"slug":"coloring-big-graphs-with-alphagozero","title":"Coloring Big Graphs with AlphaGoZero","date":"2019-02-26","arxiv_id":"1902.10162","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-multi-agent-communication-under","title":"Learning Multi-agent Communication under Limited-bandwidth Restriction for Internet Packet Routing","date":"2019-02-26","arxiv_id":"1903.05561","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-modeling-of-dense-and-incomplete","title":"Joint Modeling of Dense and Incomplete Trajectories for Citywide Traffic Volume Inference","date":"2019-02-25","arxiv_id":"1902.09255","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-deterministic-policy-with-target-for","title":"Learning Deterministic Policy with Target for Power Control in Wireless Networks","date":"2019-02-21","arxiv_id":"1902.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-generalisation-in-continuous","title":"Investigating Generalisation in Continuous Deep Reinforcement Learning","date":"2019-02-19","arxiv_id":"1902.07015","repositories_listed":0,"syntology":null},{"url":null,"slug":"message-dropout-an-efficient-training-method","title":"Message-Dropout: An Efficient Training Method for Multi-Agent Deep Reinforcement Learning","date":"2019-02-18","arxiv_id":"1902.06527","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-next-generation-airline-revenue","title":"Autonomous Airline Revenue Management: A Deep Reinforcement Learning Approach to Seat Inventory Control and Overbooking","date":"2019-02-18","arxiv_id":"1902.06824","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-topologies-between-learning","title":"Leveraging Communication Topologies Between Learning Agents in Deep Reinforcement Learning","date":"2019-02-16","arxiv_id":"1902.06740","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoqb-automl-for-network-quantization-and","title":"AutoQ: Automated Kernel-Wise Neural Network Quantization","date":"2019-02-15","arxiv_id":"1902.05690","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-high-level","title":"Deep Reinforcement Learning Based High-level Driving Behavior Decision-making Model in Heterogeneous Traffic","date":"2019-02-15","arxiv_id":"1902.05772","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-offloading-policies-for-cloud","title":"Network Offloading Policies for Cloud Robotics: a Learning-based Approach","date":"2019-02-15","arxiv_id":"1902.05703","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-perception-in-adversarial-scenarios","title":"Active Perception in Adversarial Scenarios using Maximum Entropy Deep Reinforcement Learning","date":"2019-02-14","arxiv_id":"1902.05644","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-actor-critic-in-an-ensemble","title":"Off-Policy Actor-Critic in an Ensemble: Achieving Maximum General Entropy and Effective Environment Exploration in Deep Reinforcement Learning","date":"2019-02-14","arxiv_id":"1902.05551","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-from-policy","title":"Deep Reinforcement Learning from Policy-Dependent Human Feedback","date":"2019-02-12","arxiv_id":"1902.04257","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-space-reinforcement-learning-for","title":"Latent Space Reinforcement Learning for Steering Angle Prediction","date":"2019-02-11","arxiv_id":"1902.03765","repositories_listed":0,"syntology":null},{"url":null,"slug":"wisemove-a-framework-for-safe-deep","title":"WiseMove: A Framework for Safe Deep Reinforcement Learning for Autonomous Driving","date":"2019-02-11","arxiv_id":"1902.04118","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bandit-framework-for-optimal-selection-of","title":"A Bandit Framework for Optimal Selection of Reinforcement Learning Agents","date":"2019-02-10","arxiv_id":"1902.03657","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaoptimization-on-a-distributed-system-for","title":"Metaoptimization on a Distributed System for Deep Reinforcement Learning","date":"2019-02-07","arxiv_id":"1902.02725","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-search-and-recognition-for-robot-task","title":"Visual search and recognition for robot task execution and monitoring","date":"2019-02-07","arxiv_id":"1902.02870","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-policy-distillation","title":"Distilling Policy Distillation","date":"2019-02-06","arxiv_id":"1902.02186","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-for-autonomous","title":"Adaptive Stress Testing for Autonomous Vehicles","date":"2019-02-05","arxiv_id":"1902.01909","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-learn-in-simulation","title":"Learning to Learn in Simulation","date":"2019-02-05","arxiv_id":"1902.01569","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-multimodal-multitask-learning","title":"Embodied Multimodal Multitask Learning","date":"2019-02-04","arxiv_id":"1902.01385","repositories_listed":0,"syntology":null},{"url":"/paper/joint-entity-linking-with-deep-reinforcement","slug":"joint-entity-linking-with-deep-reinforcement","title":"Joint Entity Linking with Deep Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00330","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-rationalizations-in-deep-reinforcement","title":"Visual Rationalizations in Deep Reinforcement Learning for Atari Games","date":"2019-02-01","arxiv_id":"1902.00566","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theory-of-regularized-markov-decision","title":"A Theory of Regularized Markov Decision Processes","date":"2019-01-31","arxiv_id":"1901.11275","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-in-deep-reinforcement-learning-using","title":"Transfer in Deep Reinforcement Learning Using Successor Features and Generalised Policy Improvement","date":"2019-01-30","arxiv_id":"1901.10964","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-a-multi-objective-reward-function","title":"Designing a Multi-Objective Reward Function for Creating Teams of Robotic Bodyguards Using Deep Reinforcement Learning","date":"2019-01-28","arxiv_id":"1901.09837","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-deep-reinforcement-learning-by","title":"Off-Policy Deep Reinforcement Learning by Bootstrapping the Covariate Shift","date":"2019-01-27","arxiv_id":"1901.09455","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-deep-reinforcement-learning-for","title":"Model-based Deep Reinforcement Learning for Dynamic Portfolio Optimization","date":"2019-01-25","arxiv_id":"1901.08740","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning","title":"Federated Deep Reinforcement Learning","date":"2019-01-24","arxiv_id":"1901.08277","repositories_listed":0,"syntology":null},{"url":null,"slug":"never-forget-balancing-exploration-and","title":"Never Forget: Balancing Exploration and Exploitation via Learning Optical Flow","date":"2019-01-24","arxiv_id":"1901.08486","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillation-strategies-for-proximal-policy","title":"Distillation Strategies for Proximal Policy Optimization","date":"2019-01-23","arxiv_id":"1901.08128","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-recovery-controller-for-a-quadrupedal","title":"Robust Recovery Controller for a Quadrupedal Robot using Deep Reinforcement Learning","date":"2019-01-22","arxiv_id":"1901.07517","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-retrosynthetic-planning-through-self","title":"Learning retrosynthetic planning through self-play","date":"2019-01-19","arxiv_id":"1901.06569","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionarily-curated-curriculum-learning","title":"Evolutionarily-Curated Curriculum Learning for Deep Reinforcement Learning Agents","date":"2019-01-16","arxiv_id":"1901.05431","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-sepsis-treatment-strategies-by","title":"Improving Sepsis Treatment Strategies by Combining Deep and Kernel-Based Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04670","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-tensioning-method-using-deep","title":"A New Tensioning Method using Deep Reinforcement Learning for Surgical Pattern Cutting","date":"2019-01-10","arxiv_id":"1901.03327","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-based-out-of-distribution","title":"Uncertainty-Based Out-of-Distribution Detection in Deep Reinforcement Learning","date":"2019-01-08","arxiv_id":"1901.02219","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tree-search-for-portfolio-management","title":"A* Tree Search for Portfolio Management","date":"2019-01-07","arxiv_id":"1901.01855","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-applications-of-deep-reinforcement","title":"Exploring applications of deep reinforcement learning for real-world autonomous driving systems","date":"2019-01-06","arxiv_id":"1901.01536","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-control-nets-for-deep-reinforcement","title":"Recurrent Control Nets for Deep Reinforcement Learning","date":"2019-01-06","arxiv_id":"1901.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-should-i-do-now-marrying-reinforcement","title":"What Should I Do Now? Marrying Reinforcement Learning and Symbolic Planning","date":"2019-01-06","arxiv_id":"1901.01492","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-like-autonomous-car-following-model","title":"Human-Like Autonomous Car-Following Model with Deep Reinforcement Learning","date":"2019-01-03","arxiv_id":"1901.00569","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theoretical-analysis-of-deep-q-learning","title":"A Theoretical Analysis of Deep Q-Learning","date":"2019-01-01","arxiv_id":"1901.00137","repositories_listed":0,"syntology":null},{"url":null,"slug":"complementary-reinforcement-learning-towards","title":"Complementary reinforcement learning towards explainable agents","date":"2019-01-01","arxiv_id":"1901.00188","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-multi-agent","title":"Deep Reinforcement Learning for Multi-Agent Systems: A Review of Challenges, Solutions and Applications","date":"2018-12-31","arxiv_id":"1812.11794","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-planning-networks","title":"Dynamic Planning Networks","date":"2018-12-28","arxiv_id":"1812.11240","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealing-with-limited-backhaul-capacity-in","title":"Dealing with Limited Backhaul Capacity in Millimeter Wave Systems: A Deep Reinforcement Learning Approach","date":"2018-12-27","arxiv_id":"1901.01119","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-architecture-for","title":"Quantum Adiabatic Algorithm Design using Reinforcement Learning","date":"2018-12-27","arxiv_id":"1812.10797","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-concept-of-deep-reinforcement-learning","title":"A New Concept of Deep Reinforcement Learning based Augmented General Sequence Tagging System","date":"2018-12-26","arxiv_id":"1812.10234","repositories_listed":0,"syntology":null}],"record_sha256":"88c46c6c15cdf0b8f4bbc54d1dec1d6439f42eb31abffd594129e67749a8159a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}