{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/58","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":58,"pages_in_order":59,"rows_per_page":100,"rows":[5701,5800],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/57","next":"/task/deep-reinforcement-learning/papers/59","papers":[{"url":null,"slug":"learning-unmanned-aerial-vehicle-control-for","title":"Learning Unmanned Aerial Vehicle Control for Autonomous Target Following","date":"2017-09-24","arxiv_id":"1709.08233","repositories_listed":0,"syntology":null},{"url":null,"slug":"optlayer-practical-constrained-optimization","title":"OptLayer - Practical Constrained Optimization for Deep Reinforcement Learning in the Real World","date":"2017-09-22","arxiv_id":"1709.07643","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-communication-protocols-for-learning","title":"Local Communication Protocols for Learning Complex Swarm Behaviors with Deep Reinforcement Learning","date":"2017-09-21","arxiv_id":"1709.07224","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for","title":"A Deep-Reinforcement Learning Approach for Software-Defined Networking Routing Optimization","date":"2017-09-20","arxiv_id":"1709.07080","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-dexterous","title":"Deep Reinforcement Learning for Dexterous Manipulation with Concept Networks","date":"2017-09-20","arxiv_id":"1709.06977","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-policy-learning-in-end-to-end","title":"Iterative Policy Learning in End-to-End Trainable Task-Oriented Neural Dialog Models","date":"2017-09-18","arxiv_id":"1709.06136","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-cooling-optimization-for-green","title":"Transforming Cooling Optimization for Green Data Center via Deep Reinforcement Learning","date":"2017-09-15","arxiv_id":"1709.05077","repositories_listed":0,"syntology":null},{"url":null,"slug":"shared-learning-enhancing-reinforcement-in-q","title":"Shared Learning : Enhancing Reinforcement in $Q$-Ensembles","date":"2017-09-14","arxiv_id":"1709.04909","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-personalized-human-ai-interaction","title":"Towards personalized human AI interaction - adapting the behavior of AI agents using neural signatures of subjective interest","date":"2017-09-14","arxiv_id":"1709.04574","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-ai-population-dynamics-with","title":"A Study of AI Population Dynamics with Million-agent Reinforcement Learning","date":"2017-09-13","arxiv_id":"1709.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-surrogate","title":"Deep Reinforcement Learning with Surrogate Agent-Environment Interface","date":"2017-09-12","arxiv_id":"1709.03942","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-exploit-or-listen-combining-human","title":"Explore, Exploit or Listen: Combining Human Feedback and Policy Model to Speed up Deep Reinforcement Learning in 3D Worlds","date":"2017-09-12","arxiv_id":"1709.03969","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-neural-networks-with-human","title":"Pre-training Neural Networks with Human Demonstrations for Deep Reinforcement Learning","date":"2017-09-12","arxiv_id":"1709.04083","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-quadrotor-landing-using-deep","title":"Autonomous Quadrotor Landing using Deep Reinforcement Learning","date":"2017-09-11","arxiv_id":"1709.03339","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-chatbot","title":"A Deep Reinforcement Learning Chatbot","date":"2017-09-07","arxiv_id":"1709.02349","repositories_listed":0,"syntology":null},{"url":null,"slug":"formulation-of-deep-reinforcement-learning","title":"Formulation of Deep Reinforcement Learning Architecture Toward Autonomous Driving for On-Ramp Merge","date":"2017-09-07","arxiv_id":"1709.02066","repositories_listed":0,"syntology":null},{"url":null,"slug":"book-storing-algorithm-invariant-episodes-for","title":"BOOK: Storing Algorithm-Invariant Episodes for Deep Reinforcement Learning","date":"2017-09-05","arxiv_id":"1709.01308","repositories_listed":0,"syntology":null},{"url":null,"slug":"bibi-system-description-building-with-cnns","title":"BIBI System Description: Building with CNNs and Breaking with Deep Reinforcement Learning","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-mechanism-design-for-e-commerce","title":"Reinforcement Mechanism Design for e-commerce","date":"2017-08-25","arxiv_id":"1708.07607","repositories_listed":0,"syntology":null},{"url":null,"slug":"fake-news-in-social-networks","title":"Fake News in Social Networks","date":"2017-08-21","arxiv_id":"1708.06233","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-q-network-for-the-beer-game-a-deep","title":"A Deep Q-Network for the Beer Game: A Deep Reinforcement Learning algorithm to Solve Inventory Optimization Problems","date":"2017-08-20","arxiv_id":"1708.05924","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-a-new-3d-bin-packing-problem-with","title":"Solving a New 3D Bin Packing Problem with Deep Reinforcement Learning Method","date":"2017-08-20","arxiv_id":"1708.05930","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brief-survey-of-deep-reinforcement-learning","title":"A Brief Survey of Deep Reinforcement Learning","date":"2017-08-19","arxiv_id":"1708.05866","repositories_listed":0,"syntology":null},{"url":null,"slug":"ladder-a-human-level-bidding-agent-for-large","title":"LADDER: A Human-Level Bidding Agent for Large-Scale Real-Time Online Auctions","date":"2017-08-18","arxiv_id":"1708.05565","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-high","title":"Deep Reinforcement Learning for High Precision Assembly Tasks","date":"2017-08-14","arxiv_id":"1708.04033","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-machine-learning-approach-to-routing","title":"A Machine Learning Approach to Routing","date":"2017-08-10","arxiv_id":"1708.03074","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-aware-face-hallucination-via-deep","title":"Attention-Aware Face Hallucination via Deep Reinforcement Learning","date":"2017-08-10","arxiv_id":"1708.03132","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-optimality-principle","title":"An Information-Theoretic Optimality Principle for Deep Reinforcement Learning","date":"2017-08-06","arxiv_id":"1708.01867","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-inquiry","title":"Deep Reinforcement Learning for Inquiry Dialog Policies with Logical Formula Embeddings","date":"2017-08-02","arxiv_id":"1708.00667","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-stochastic-policy-gradients-in","title":"Improving Stochastic Policy Gradients in Continuous Control with Deep Reinforcement Learning using the Beta Distribution","date":"2017-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-load-control-of-thermostatically","title":"Direct Load Control of Thermostatically Controlled Loads Based on Sparse Observations Using Deep Reinforcement Learning","date":"2017-07-26","arxiv_id":"1707.08553","repositories_listed":0,"syntology":null},{"url":null,"slug":"3dcnn-dqn-rnn-a-deep-reinforcement-learning","title":"3DCNN-DQN-RNN: A Deep Reinforcement Learning Framework for Semantic Parsing of Large-scale 3D Point Clouds","date":"2017-07-21","arxiv_id":"1707.06783","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-line-building-energy-optimization-using","title":"On-line Building Energy Optimization using Deep Reinforcement Learning","date":"2017-07-18","arxiv_id":"1707.05878","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-as-online-decision-making-learning-a","title":"Tracking as Online Decision-Making: Learning a Policy from Streaming Videos with Reinforcement Learning","date":"2017-07-17","arxiv_id":"1707.04991","repositories_listed":0,"syntology":null},{"url":null,"slug":"distral-robust-multitask-reinforcement","title":"Distral: Robust Multitask Reinforcement Learning","date":"2017-07-13","arxiv_id":"1707.04175","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-macromanagement-in-starcraft-from","title":"Learning Macromanagement in StarCraft from Replays using Deep Learning","date":"2017-07-12","arxiv_id":"1707.03743","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-learning-for-self-organizing-networks","title":"Deep Q-Learning for Self-Organizing Networks Fault Management and Radio Performance Improvement","date":"2017-07-10","arxiv_id":"1707.02329","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-attention","title":"Deep Reinforcement Learning Attention Selection for Person Re-Identification","date":"2017-07-10","arxiv_id":"1707.02785","repositories_listed":0,"syntology":null},{"url":null,"slug":"hashing-over-predicted-future-frames-for","title":"Hashing over Predicted Future Frames for Informed Exploration of Deep Reinforcement Learning","date":"2017-07-03","arxiv_id":"1707.00524","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-actor-critic-reinforcement","title":"Sample-efficient Actor-Critic Reinforcement Learning with Supervised Data for Dialogue Management","date":"2017-07-01","arxiv_id":"1707.00130","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-learning-in-motor-controla-deep","title":"Structure Learning in Motor Control:A Deep Reinforcement Learning Model","date":"2017-06-21","arxiv_id":"1706.06827","repositories_listed":0,"syntology":null},{"url":null,"slug":"ucb-exploration-via-q-ensembles","title":"UCB Exploration via Q-Ensembles","date":"2017-06-05","arxiv_id":"1706.01502","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpolated-policy-gradient-merging-on","title":"Interpolated Policy Gradient: Merging On-Policy and Off-Policy Gradient Estimation for Deep Reinforcement Learning","date":"2017-06-01","arxiv_id":"1706.00387","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-learning-rate","title":"Reinforcement Learning for Learning Rate Control","date":"2017-05-31","arxiv_id":"1705.11159","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-active-object-tracking-via","title":"End-to-end Active Object Tracking via Reinforcement Learning","date":"2017-05-30","arxiv_id":"1705.10561","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-acceleration-control-for","title":"Fine-grained acceleration control for autonomous intersection management using deep reinforcement learning","date":"2017-05-30","arxiv_id":"1705.10432","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-perceptual-reward-functions","title":"Cross-Domain Perceptual Reward Functions","date":"2017-05-25","arxiv_id":"1705.09045","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-space-decomposition-and-subgoal","title":"State Space Decomposition and Subgoal Creation for Transfer in Deep Reinforcement Learning","date":"2017-05-24","arxiv_id":"1705.08997","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-state-space-models-for-optimal","title":"Continuous State-Space Models for Optimal Sepsis Treatment - a Deep Reinforcement Learning Approach","date":"2017-05-23","arxiv_id":"1705.08422","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-experience-replay-generation-for","title":"Enhanced Experience Replay Generation for Efficient Reinforcement Learning","date":"2017-05-23","arxiv_id":"1705.08245","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-mix-n-step-returns-generalizing","title":"Learning to Mix n-Step Returns: Generalizing lambda-Returns for Deep Reinforcement Learning","date":"2017-05-21","arxiv_id":"1705.07445","repositories_listed":0,"syntology":null},{"url":null,"slug":"shallow-updates-for-deep-reinforcement","title":"Shallow Updates for Deep Reinforcement Learning","date":"2017-05-21","arxiv_id":"1705.07461","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-factor-policies-and-action-value","title":"Learning to Factor Policies and Action-Value Functions: Factored Action Space Representations for Deep Reinforcement learning","date":"2017-05-20","arxiv_id":"1705.07269","repositories_listed":0,"syntology":null},{"url":null,"slug":"atari-games-and-intel-processors","title":"Atari games and Intel processors","date":"2017-05-19","arxiv_id":"1705.06936","repositories_listed":0,"syntology":null},{"url":null,"slug":"delving-into-adversarial-attacks-on-deep","title":"Delving into adversarial attacks on deep policies","date":"2017-05-18","arxiv_id":"1705.06452","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-knowledge-transfer-in-deep-q","title":"Analyzing Knowledge Transfer in Deep Q-Networks for Autonomously Handling Multiple Intersections","date":"2017-05-02","arxiv_id":"1705.01197","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-occluded-intersections-with","title":"Navigating Occluded Intersections with Autonomous Vehicles using Deep Reinforcement Learning","date":"2017-05-02","arxiv_id":"1705.01196","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-image","title":"Deep Reinforcement Learning-based Image Captioning with Embedding Reward","date":"2017-04-12","arxiv_id":"1704.03899","repositories_listed":0,"syntology":null},{"url":null,"slug":"composite-task-completion-dialogue-policy","title":"Composite Task-Completion Dialogue Policy Learning via Hierarchical Deep Reinforcement Learning","date":"2017-04-10","arxiv_id":"1704.03084","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-deep-reinforcement-learning","title":"Data-efficient Deep Reinforcement Learning for Dexterous Manipulation","date":"2017-04-10","arxiv_id":"1704.03073","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-persuasion-strategies-and-deep","title":"Evaluating Persuasion Strategies and Deep Reinforcement Learning methods for Negotiation Dialogue agents","date":"2017-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-approach-to-natural-language","title":"An End-to-End Approach to Natural Language Object Retrieval via Context-Aware Deep Reinforcement Learning","date":"2017-03-22","arxiv_id":"1703.07579","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-framework-of-cloud-resource","title":"A Hierarchical Framework of Cloud Resource Allocation and Power Management Using Deep Reinforcement Learning","date":"2017-03-13","arxiv_id":"1703.04221","repositories_listed":0,"syntology":null},{"url":null,"slug":"sensor-fusion-for-robot-control-through-deep","title":"Sensor Fusion for Robot Control through Deep Reinforcement Learning","date":"2017-03-13","arxiv_id":"1703.04550","repositories_listed":0,"syntology":null},{"url":null,"slug":"micro-objective-learning-accelerating-deep","title":"Micro-Objective Learning : Accelerating Deep Reinforcement Learning through the Discovery of Continuous Subgoals","date":"2017-03-11","arxiv_id":"1703.03933","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactics-of-adversarial-attack-on-deep","title":"Tactics of Adversarial Attack on Deep Reinforcement Learning Agents","date":"2017-03-08","arxiv_id":"1703.06748","repositories_listed":0,"syntology":null},{"url":null,"slug":"surprise-based-intrinsic-motivation-for-deep","title":"Surprise-Based Intrinsic Motivation for Deep Reinforcement Learning","date":"2017-03-06","arxiv_id":"1703.01732","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-what-data-to-learn","title":"Learning What Data to Learn","date":"2017-02-28","arxiv_id":"1702.08635","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-for-air-hockey-striking","title":"Learning Control for Air Hockey Striking using Deep Reinforcement Learning","date":"2017-02-26","arxiv_id":"1702.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-meta-learning-by-parallel-algorithm","title":"Online Meta-learning by Parallel Algorithm Competition","date":"2017-02-24","arxiv_id":"1702.07490","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-gains-social-intelligence-through","title":"Robot gains Social Intelligence through Multimodal Deep Reinforcement Learning","date":"2017-02-24","arxiv_id":"1702.07492","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-repeat-fine-grained-action","title":"Learning to Repeat: Fine Grained Action Repetition for Deep Reinforcement Learning","date":"2017-02-20","arxiv_id":"1702.06054","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-deep-reinforcement-learning-for-1","title":"Collaborative Deep Reinforcement Learning for Joint Object Search","date":"2017-02-18","arxiv_id":"1702.05573","repositories_listed":0,"syntology":null},{"url":"/paper/sigmoid-weighted-linear-units-for-neural","slug":"sigmoid-weighted-linear-units-for-neural","title":"Sigmoid-Weighted Linear Units for Neural Network Function Approximation in Reinforcement Learning","date":"2017-02-10","arxiv_id":"1702.03118","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robotic-1","title":"Deep Reinforcement Learning for Robotic Manipulation-The state of the art","date":"2017-01-31","arxiv_id":"1701.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-visual-object","title":"Deep Reinforcement Learning for Visual Object Tracking in Videos","date":"2017-01-31","arxiv_id":"1701.08936","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-level-control-of-ramp-metering-based","title":"Expert Level control of Ramp Metering based on Multi-task Deep Reinforcement Learning","date":"2017-01-30","arxiv_id":"1701.08832","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-deterministic-policy-improvement","title":"Non-Deterministic Policy Improvement Stabilizes Approximated Reinforcement Learning","date":"2016-12-22","arxiv_id":"1612.07548","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-deep-reinforcement-learning-1","title":"Sample-efficient Deep Reinforcement Learning for Dialog Control","date":"2016-12-18","arxiv_id":"1612.06000","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-successor","title":"Deep Reinforcement Learning with Successor Features for Navigation across Similar Environments","date":"2016-12-16","arxiv_id":"1612.05533","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-deep-reinforcement-learning-for","title":"End-to-End Deep Reinforcement Learning for Lane Keeping Assist","date":"2016-12-13","arxiv_id":"1612.04340","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-skills-with-semi-supervised","title":"Generalizing Skills with Semi-Supervised Reinforcement Learning","date":"2016-12-01","arxiv_id":"1612.00429","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-feature-encoding-for-reinforcement","title":"Linear Feature Encoding for Reinforcement Learning","date":"2016-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"averaged-dqn-variance-reduction-and","title":"Averaged-DQN: Variance Reduction and Stabilization for Deep Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.01929","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-perform-physics-experiments-via","title":"Learning to Perform Physics Experiments via Deep Reinforcement Learning","date":"2016-11-06","arxiv_id":"1611.01843","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-with-deep-model-based","title":"Multi-task learning with deep model based reinforcement learning","date":"2016-11-04","arxiv_id":"1611.01457","repositories_listed":0,"syntology":null},{"url":null,"slug":"combating-reinforcement-learnings-sisyphean","title":"Combating Reinforcement Learning's Sisyphean Curse with Intrinsic Fear","date":"2016-11-03","arxiv_id":"1611.01211","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-locomotion-skills-using-deeprl-does","title":"Learning Locomotion Skills Using DeepRL: Does the Choice of Action Space Matter?","date":"2016-11-03","arxiv_id":"1611.01055","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-a-deep-reinforcement-learning-agent-for","title":"Using a Deep Reinforcement Learning Agent for Traffic Signal Control","date":"2016-11-03","arxiv_id":"1611.01142","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-runtime-parameters-in-computer","title":"Learning Runtime Parameters in Computer Systems with Delayed Experience Injection","date":"2016-10-31","arxiv_id":"1610.09903","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-deep-q-networks-for-sim-to-real","title":"Modular Deep Q Networks for Sim-to-real Transfer of Visuo-motor Policies","date":"2016-10-21","arxiv_id":"1610.06781","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilization-of-deep-reinforcement-learning","title":"Utilization of Deep Reinforcement Learning for saccadic-based object visual search","date":"2016-10-20","arxiv_id":"1610.06492","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-contrastive-divergence-with-generative","title":"Online Contrastive Divergence with Generative Replay: Experience Replay without Storing Data","date":"2016-10-18","arxiv_id":"1610.05555","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-robot-learning-from-pixels-with","title":"Sim-to-Real Robot Learning from Pixels with Progressive Nets","date":"2016-10-13","arxiv_id":"1610.04286","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multi-agent-reinforcement-learning-for","title":"Safe, Multi-Agent, Reinforcement Learning for Autonomous Driving","date":"2016-10-11","arxiv_id":"1610.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-from-raw-pixels","title":"Deep Reinforcement Learning From Raw Pixels in Doom","date":"2016-10-07","arxiv_id":"1610.02164","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cognitive-exploration-through-deep","title":"Towards Cognitive Exploration through Deep Reinforcement Learning for Mobile Robots","date":"2016-10-06","arxiv_id":"1610.01733","repositories_listed":0,"syntology":null},{"url":null,"slug":"reset-free-guided-policy-search-efficient","title":"Reset-Free Guided Policy Search: Efficient Deep Reinforcement Learning with Stochastic Initial States","date":"2016-10-04","arxiv_id":"1610.01112","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robotic","title":"Deep Reinforcement Learning for Robotic Manipulation with Asynchronous Off-Policy Updates","date":"2016-10-03","arxiv_id":"1610.00633","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-tensegrity","title":"Deep Reinforcement Learning for Tensegrity Robot Locomotion","date":"2016-09-28","arxiv_id":"1609.09049","repositories_listed":0,"syntology":null}],"record_sha256":"86bf4603fffcf31bbd59fa51f6211fe3f76a3faf90593ae07e9081bdc5fd91b8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}