{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/148","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":148,"pages_in_order":152,"rows_per_page":100,"rows":[14701,14800],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/147","next":"/task/reinforcement-learning-1/papers/149","papers":[{"url":null,"slug":"show-attend-and-interact-perceivable-human","title":"Show, Attend and Interact: Perceivable Human-Robot Social Interaction through Neural Attention Q-Network","date":"2017-02-28","arxiv_id":"1702.08626","repositories_listed":0,"syntology":null},{"url":"/paper/a-dataset-for-developing-and-benchmarking","slug":"a-dataset-for-developing-and-benchmarking","title":"A Dataset for Developing and Benchmarking Active Vision","date":"2017-02-27","arxiv_id":"1702.08272","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-for-air-hockey-striking","title":"Learning Control for Air Hockey Striking using Deep Reinforcement Learning","date":"2017-02-26","arxiv_id":"1702.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-variance-reduction-methods-for","title":"Stochastic Variance Reduction Methods for Policy Evaluation","date":"2017-02-25","arxiv_id":"1702.07944","repositories_listed":0,"syntology":null},{"url":null,"slug":"changing-model-behavior-at-test-time-using","title":"Changing Model Behavior at Test-Time Using Reinforcement Learning","date":"2017-02-24","arxiv_id":"1702.07780","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-gene-regulatory-networks-with","title":"Control of Gene Regulatory Networks with Noisy Measurements and Uncertain Inputs","date":"2017-02-24","arxiv_id":"1702.07652","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-meta-learning-by-parallel-algorithm","title":"Online Meta-learning by Parallel Algorithm Competition","date":"2017-02-24","arxiv_id":"1702.07490","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-gains-social-intelligence-through","title":"Robot gains Social Intelligence through Multimodal Deep Reinforcement Learning","date":"2017-02-24","arxiv_id":"1702.07492","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-representation-for-lifetime-value","title":"Automatic Representation for Lifetime Value Recommender Systems","date":"2017-02-23","arxiv_id":"1702.07125","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-distillation-for-controlling-specificity","title":"Data Distillation for Controlling Specificity in Dialogue Generation","date":"2017-02-22","arxiv_id":"1702.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-argument","title":"Reinforcement Learning Based Argument Component Detection","date":"2017-02-21","arxiv_id":"1702.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-repeat-fine-grained-action","title":"Learning to Repeat: Fine Grained Action Repetition for Deep Reinforcement Learning","date":"2017-02-20","arxiv_id":"1702.06054","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-deep-reinforcement-learning-for-1","title":"Collaborative Deep Reinforcement Learning for Joint Object Search","date":"2017-02-18","arxiv_id":"1702.05573","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-policy-gradient-methods-for-improving","title":"Batch Policy Gradient Methods for Improving Neural Conversation Models","date":"2017-02-10","arxiv_id":"1702.03334","repositories_listed":0,"syntology":null},{"url":"/paper/sigmoid-weighted-linear-units-for-neural","slug":"sigmoid-weighted-linear-units-for-neural","title":"Sigmoid-Weighted Linear Units for Neural Network Function Approximation in Reinforcement Learning","date":"2017-02-10","arxiv_id":"1702.03118","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-qa-with-generative-domain","title":"Semi-Supervised QA with Generative Domain-Adaptive Nets","date":"2017-02-07","arxiv_id":"1702.02206","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-reinforcement-learning-for","title":"Uncertainty-Aware Reinforcement Learning for Collision Avoidance","date":"2017-02-03","arxiv_id":"1702.01182","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robotic-1","title":"Deep Reinforcement Learning for Robotic Manipulation-The state of the art","date":"2017-01-31","arxiv_id":"1701.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-visual-object","title":"Deep Reinforcement Learning for Visual Object Tracking in Videos","date":"2017-01-31","arxiv_id":"1701.08936","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-level-control-of-ramp-metering-based","title":"Expert Level control of Ramp Metering based on Multi-task Deep Reinforcement Learning","date":"2017-01-30","arxiv_id":"1701.08832","repositories_listed":0,"syntology":null},{"url":null,"slug":"flow-navigation-by-smart-microswimmers-via","title":"Flow Navigation by Smart Microswimmers via Reinforcement Learning","date":"2017-01-30","arxiv_id":"1701.08848","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-algorithm-selection","title":"Reinforcement Learning Algorithm Selection","date":"2017-01-30","arxiv_id":"1701.08810","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-approaches-to-ucav","title":"Artificial Intelligence Approaches To UCAV Autonomy","date":"2017-01-24","arxiv_id":"1701.07103","repositories_listed":0,"syntology":null},{"url":null,"slug":"binary-matrix-guessing-problem","title":"Binary Matrix Guessing Problem","date":"2017-01-22","arxiv_id":"1701.06167","repositories_listed":0,"syntology":null},{"url":null,"slug":"basic-protocols-in-quantum-reinforcement","title":"Basic protocols in quantum reinforcement learning with superconducting circuits","date":"2017-01-18","arxiv_id":"1701.05131","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-agnostic-human-in-the-loop","title":"Agent-Agnostic Human-in-the-Loop Reinforcement Learning","date":"2017-01-15","arxiv_id":"1701.04079","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-and-incremental-learning-of-gaussian","title":"Scalable and Incremental Learning of Gaussian Mixture Models","date":"2017-01-14","arxiv_id":"1701.03940","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-embodied-agents","title":"Reinforcement Learning based Embodied Agents Modelling Human Users Through Interaction and Multi-Sensory Perception","date":"2017-01-09","arxiv_id":"1701.02369","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-neural-network-based-machine","title":"A Review of Neural Network Based Machine Learning Approaches for Rotor Angle Stability Control","date":"2017-01-05","arxiv_id":"1701.01214","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-negotiable-reinforcement-learning","title":"Toward negotiable reinforcement learning: shifting priorities in Pareto optimal sequential decision-making","date":"2017-01-05","arxiv_id":"1701.01302","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-person-activity-forecasting-with-online","title":"First-Person Activity Forecasting with Online Inverse Reinforcement Learning","date":"2016-12-22","arxiv_id":"1612.07796","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-deterministic-policy-improvement","title":"Non-Deterministic Policy Improvement Stabilizes Approximated Reinforcement Learning","date":"2016-12-22","arxiv_id":"1612.07548","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-function-approximation-error-for-risk","title":"On the function approximation error for risk-sensitive reinforcement learning","date":"2016-12-22","arxiv_id":"1612.07562","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-is-its-own-reward-self-supervision-for","title":"Loss is its own Reward: Self-Supervision for Reinforcement Learning","date":"2016-12-21","arxiv_id":"1612.07307","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-perceptual-rewards-for-imitation","title":"Unsupervised Perceptual Rewards for Imitation Learning","date":"2016-12-20","arxiv_id":"1612.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-deep-reinforcement-learning-1","title":"Sample-efficient Deep Reinforcement Learning for Dialog Control","date":"2016-12-18","arxiv_id":"1612.06000","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-predict-where-to-look-in","title":"Learning to predict where to look in interactive environments using deep recurrent q-learning","date":"2016-12-17","arxiv_id":"1612.05753","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-quantum","title":"Reinforcement Learning Using Quantum Boltzmann Machines","date":"2016-12-17","arxiv_id":"1612.05695","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-successor","title":"Deep Reinforcement Learning with Successor Features for Navigation across Similar Environments","date":"2016-12-16","arxiv_id":"1612.05533","repositories_listed":0,"syntology":null},{"url":null,"slug":"separation-of-concerns-in-reinforcement","title":"Separation of Concerns in Reinforcement Learning","date":"2016-12-15","arxiv_id":"1612.05159","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-deep-reinforcement-learning-for","title":"End-to-End Deep Reinforcement Learning for Lane Keeping Assist","date":"2016-12-13","arxiv_id":"1612.04340","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-human-domain-knowledge-into","title":"Incorporating Human Domain Knowledge into Large Scale Cost Function Learning","date":"2016-12-13","arxiv_id":"1612.04318","repositories_listed":0,"syntology":null},{"url":null,"slug":"response-to-comment-on-perceptual-learning","title":"Response to Comment on 'Perceptual Learning Incepted by Decoded fMRI Neurofeedback Without Stimulus Presentation'; How can a decoded neurofeedback method (DecNef) lead to successful reinforcement and visual perceptual learning?","date":"2016-12-13","arxiv_id":"1612.04234","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-drive-using-inverse-reinforcement","title":"Learning to Drive using Inverse Reinforcement Learning and Deep Q-Networks","date":"2016-12-12","arxiv_id":"1612.03653","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-for-real-time","title":"Online Reinforcement Learning for Real-Time Exploration in Continuous State and Action Markov Decision Processes","date":"2016-12-12","arxiv_id":"1612.03780","repositories_listed":0,"syntology":null},{"url":null,"slug":"poseagent-budget-constrained-6d-object-pose","title":"PoseAgent: Budget-Constrained 6D Object Pose Estimation via Reinforcement Learning","date":"2016-12-12","arxiv_id":"1612.03779","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-temporal-logic","title":"Reinforcement Learning With Temporal Logic Rewards","date":"2016-12-11","arxiv_id":"1612.03471","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-deep-learning-with-spiking-neurons-in","title":"Towards deep learning with spiking neurons in energy based models with contrastive Hebbian plasticity","date":"2016-12-09","arxiv_id":"1612.03214","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchy-through-composition-with-linearly","title":"Hierarchy through Composition with Linearly Solvable Markov Decision Processes","date":"2016-12-08","arxiv_id":"1612.02757","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-primal-dual-methods-and-sample","title":"Stochastic Primal-Dual Methods and Sample Complexity of Reinforcement Learning","date":"2016-12-08","arxiv_id":"1612.02516","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-information-seeking-agents","title":"Towards Information-Seeking Agents","date":"2016-12-08","arxiv_id":"1612.02605","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-of-robotic-tasks-without-a","title":"Deep Learning of Robotic Tasks without a Simulator using Strong and Weak Human Supervision","date":"2016-12-04","arxiv_id":"1612.01086","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-superoptimize-programs-workshop","title":"Learning to superoptimize programs - Workshop Version","date":"2016-12-04","arxiv_id":"1612.01094","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-optimal-training-of-animal-behavior","title":"Adaptive optimal training of animal behavior","date":"2016-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-incremental-dialogue-systems","title":"Bootstrapping incremental dialogue systems: using linguistic knowledge to learn from minimal data","date":"2016-12-01","arxiv_id":"1612.00347","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-skills-with-semi-supervised","title":"Generalizing Skills with Semi-Supervised Reinforcement Learning","date":"2016-12-01","arxiv_id":"1612.00429","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-feature-encoding-for-reinforcement","title":"Linear Feature Encoding for Reinforcement Learning","date":"2016-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"showing-versus-doing-teaching-by","title":"Showing versus doing: Teaching by demonstration","date":"2016-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-for-multi-task-reinforcement","title":"Exploration for Multi-task Reinforcement Learning with Deep Generative Models","date":"2016-11-29","arxiv_id":"1611.09894","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-policy-gradient-by-exploring-under","title":"Improving Policy Gradient by Exploring Under-appreciated Rewards","date":"2016-11-28","arxiv_id":"1611.09321","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-compose-words-into-sentences-with","title":"Learning to Compose Words into Sentences with Reinforcement Learning","date":"2016-11-28","arxiv_id":"1611.09100","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonparametric-general-reinforcement-learning","title":"Nonparametric General Reinforcement Learning","date":"2016-11-28","arxiv_id":"1611.08944","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiscale-inverse-reinforcement-learning","title":"Multiscale Inverse Reinforcement Learning using Diffusion Wavelets","date":"2016-11-24","arxiv_id":"1611.08070","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-attention-models-for-depth-based","title":"Recurrent Attention Models for Depth-Based Person Identification","date":"2016-11-22","arxiv_id":"1611.07212","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-learning-approach-for-joint-video","title":"A Deep Learning Approach for Joint Video Frame and Reward Prediction in Atari Games","date":"2016-11-21","arxiv_id":"1611.07078","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-lens-how-much-memory-does-an-agent-use","title":"Memory Lens: How Much Memory Does an Agent Use?","date":"2016-11-21","arxiv_id":"1611.06928","repositories_listed":0,"syntology":null},{"url":null,"slug":"options-discovery-with-budgeted-reinforcement","title":"Options Discovery with Budgeted Reinforcement Learning","date":"2016-11-21","arxiv_id":"1611.06824","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-rich-observation","title":"Reinforcement Learning in Rich-Observation MDPs using Spectral Methods","date":"2016-11-11","arxiv_id":"1611.03907","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-reinforcement-learning","title":"Fairness in Reinforcement Learning","date":"2016-11-09","arxiv_id":"1611.03071","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-tutor-conservative-fine-tuning-of","title":"Sequence Tutor: Conservative Fine-Tuning of Sequence Generation Models with KL-control","date":"2016-11-09","arxiv_id":"1611.02796","repositories_listed":0,"syntology":null},{"url":null,"slug":"averaged-dqn-variance-reduction-and","title":"Averaged-DQN: Variance Reduction and Stabilization for Deep Reinforcement Learning","date":"2016-11-07","arxiv_id":"1611.01929","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for","title":"Reinforcement Learning Approach for Parallelization in Filters Aggregation Based Feature Selection Algorithms","date":"2016-11-07","arxiv_id":"1611.02047","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-perform-physics-experiments-via","title":"Learning to Perform Physics Experiments via Deep Reinforcement Learning","date":"2016-11-06","arxiv_id":"1611.01843","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-with-deep-model-based","title":"Multi-task learning with deep model based reinforcement learning","date":"2016-11-04","arxiv_id":"1611.01457","repositories_listed":0,"syntology":null},{"url":null,"slug":"combating-reinforcement-learnings-sisyphean","title":"Combating Reinforcement Learning's Sisyphean Curse with Intrinsic Fear","date":"2016-11-03","arxiv_id":"1611.01211","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-locomotion-skills-using-deeprl-does","title":"Learning Locomotion Skills Using DeepRL: Does the Choice of Action Space Matter?","date":"2016-11-03","arxiv_id":"1611.01055","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantile-reinforcement-learning","title":"Quantile Reinforcement Learning","date":"2016-11-03","arxiv_id":"1611.00862","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-a-deep-reinforcement-learning-agent-for","title":"Using a Deep Reinforcement Learning Agent for Traffic Signal Control","date":"2016-11-03","arxiv_id":"1611.01142","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-runtime-parameters-in-computer","title":"Learning Runtime Parameters in Computer Systems with Delayed Experience Injection","date":"2016-10-31","arxiv_id":"1610.09903","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-decision-processes-with-low","title":"Contextual Decision Processes with Low Bellman Rank are PAC-Learnable","date":"2016-10-29","arxiv_id":"1610.09512","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-enhanced-machine-learning","title":"Quantum-enhanced machine learning","date":"2016-10-26","arxiv_id":"1610.08251","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-conflicting","title":"Reinforcement Learning in Conflicting Environments for Autonomous Vehicles","date":"2016-10-22","arxiv_id":"1610.07089","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilization-of-deep-reinforcement-learning","title":"Utilization of Deep Reinforcement Learning for saccadic-based object visual search","date":"2016-10-20","arxiv_id":"1610.06492","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-to-the-view","title":"A Reinforcement Learning Approach to the View Planning Problem","date":"2016-10-19","arxiv_id":"1610.06204","repositories_listed":0,"syntology":null},{"url":null,"slug":"particle-swarm-optimization-for-generating","title":"Particle Swarm Optimization for Generating Interpretable Fuzzy Reinforcement Learning Policies","date":"2016-10-19","arxiv_id":"1610.05984","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-contrastive-divergence-with-generative","title":"Online Contrastive Divergence with Generative Replay: Experience Replay without Storing Data","date":"2016-10-18","arxiv_id":"1610.05555","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-end-of-optimism-an-asymptotic-analysis-of","title":"The End of Optimism? An Asymptotic Analysis of Finite-Armed Linear Bandits","date":"2016-10-14","arxiv_id":"1610.04491","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-robot-learning-from-pixels-with","title":"Sim-to-Real Robot Learning from Pixels with Progressive Nets","date":"2016-10-13","arxiv_id":"1610.04286","repositories_listed":0,"syntology":null},{"url":null,"slug":"introduction-to-the-industrial-benchmark","title":"Introduction to the \"Industrial Benchmark\"","date":"2016-10-12","arxiv_id":"1610.03793","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigational-instruction-generation-as","title":"Navigational Instruction Generation as Inverse Reinforcement Learning with Neural Machine Translation","date":"2016-10-11","arxiv_id":"1610.03164","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multi-agent-reinforcement-learning-for","title":"Safe, Multi-Agent, Reinforcement Learning for Autonomous Driving","date":"2016-10-11","arxiv_id":"1610.03295","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalizing-a-dialogue-system-with-transfer","title":"Personalizing a Dialogue System with Transfer Reinforcement Learning","date":"2016-10-10","arxiv_id":"1610.02891","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-from-raw-pixels","title":"Deep Reinforcement Learning From Raw Pixels in Doom","date":"2016-10-07","arxiv_id":"1610.02164","repositories_listed":0,"syntology":null},{"url":null,"slug":"connecting-generative-adversarial-networks","title":"Connecting Generative Adversarial Networks and Actor-Critic Methods","date":"2016-10-06","arxiv_id":"1610.01945","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cognitive-exploration-through-deep","title":"Towards Cognitive Exploration through Deep Reinforcement Learning for Mobile Robots","date":"2016-10-06","arxiv_id":"1610.01733","repositories_listed":0,"syntology":null},{"url":null,"slug":"reset-free-guided-policy-search-efficient","title":"Reset-Free Guided Policy Search: Efficient Deep Reinforcement Learning with Stochastic Initial States","date":"2016-10-04","arxiv_id":"1610.01112","repositories_listed":0,"syntology":null},{"url":null,"slug":"collective-robot-reinforcement-learning-with","title":"Collective Robot Reinforcement Learning with Distributed Asynchronous Guided Policy Search","date":"2016-10-03","arxiv_id":"1610.00673","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robotic","title":"Deep Reinforcement Learning for Robotic Manipulation with Asynchronous Off-Policy Updates","date":"2016-10-03","arxiv_id":"1610.00633","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-tensegrity","title":"Deep Reinforcement Learning for Tensegrity Robot Locomotion","date":"2016-09-28","arxiv_id":"1609.09049","repositories_listed":0,"syntology":null},{"url":null,"slug":"ubuntuworld-10-lts-a-platform-for-automated","title":"UbuntuWorld 1.0 LTS - A Platform for Automated Problem Solving & Troubleshooting in the Ubuntu OS","date":"2016-09-27","arxiv_id":"1609.08524","repositories_listed":0,"syntology":null}],"record_sha256":"23139fd13186bb363fd842d17c1fc963aaa30a7f0ac2c042e6f36ba5503fdc72","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}