{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/127","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":127,"pages_in_order":132,"rows_per_page":100,"rows":[12601,12700],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/126","next":"/task/reinforcement-learning/papers/128","papers":[{"url":null,"slug":"inverse-risk-sensitive-reinforcement-learning","title":"Inverse Risk-Sensitive Reinforcement Learning","date":"2017-03-29","arxiv_id":"1703.09842","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-from-summary","title":"Inverse Reinforcement Learning from Summary Data","date":"2017-03-28","arxiv_id":"1703.09700","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-for-linear-quadratic","title":"Thompson Sampling for Linear-Quadratic Control Problems","date":"2017-03-27","arxiv_id":"1703.08972","repositories_listed":0,"syntology":null},{"url":null,"slug":"cohesion-based-online-actor-critic","title":"Cohesion-based Online Actor-Critic Reinforcement Learning for mHealth Intervention","date":"2017-03-25","arxiv_id":"1703.10039","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-exploitation-in-mdps-with-options","title":"Exploration--Exploitation in MDPs with Options","date":"2017-03-25","arxiv_id":"1703.08667","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-discovery-of-deep-options","title":"Multi-Level Discovery of Deep Options","date":"2017-03-24","arxiv_id":"1703.08294","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-basis-function-adaptation-for-1","title":"Unsupervised Basis Function Adaptation for Reinforcement Learning","date":"2017-03-23","arxiv_id":"1703.07940","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-approach-to-natural-language","title":"An End-to-End Approach to Natural Language Object Retrieval via Context-Aware Deep Reinforcement Learning","date":"2017-03-22","arxiv_id":"1703.07579","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-exploration-via-randomized-value","title":"Deep Exploration via Randomized Value Functions","date":"2017-03-22","arxiv_id":"1703.07608","repositories_listed":0,"syntology":null},{"url":null,"slug":"fake-news-mitigation-via-point-process-based","title":"Fake News Mitigation via Point Process Based Intervention","date":"2017-03-22","arxiv_id":"1703.07823","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-model-identification","title":"Information-theoretic Model Identification and Policy Search using Physics Engines with Application to Robotic Manipulation","date":"2017-03-22","arxiv_id":"1703.07822","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-language-understanding","title":"Investigation of Language Understanding Impact for Reinforcement Learning Based Dialogue Systems","date":"2017-03-21","arxiv_id":"1703.07055","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudorehearsal-in-value-function","title":"Pseudorehearsal in value function approximation","date":"2017-03-21","arxiv_id":"1703.07075","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-timescale-gradient-descent-temporal","title":"Multi-Timescale, Gradient Descent, Temporal Difference Learning with Linear Options","date":"2017-03-19","arxiv_id":"1703.06471","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-decentralized-multi-task-multi-agent","title":"Deep Decentralized Multi-task Multi-Agent Reinforcement Learning under Partial Observability","date":"2017-03-17","arxiv_id":"1703.06182","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-learning-for-offloading-and","title":"Online Learning for Offloading and Autoscaling in Energy Harvesting Mobile Edge Computing","date":"2017-03-17","arxiv_id":"1703.06060","repositories_listed":0,"syntology":null},{"url":null,"slug":"particle-value-functions","title":"Particle Value Functions","date":"2017-03-16","arxiv_id":"1703.05820","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-for-demand","title":"Using Reinforcement Learning for Demand Response of Domestic Hot Water Buffers: a Real-Life Demonstration","date":"2017-03-16","arxiv_id":"1703.05486","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-of-two-timescale","title":"Finite Sample Analysis of Two-Timescale Stochastic Approximation with Applications to Reinforcement Learning","date":"2017-03-15","arxiv_id":"1703.05376","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-framework-of-cloud-resource","title":"A Hierarchical Framework of Cloud Resource Allocation and Power Management Using Deep Reinforcement Learning","date":"2017-03-13","arxiv_id":"1703.04221","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-transition-based","title":"Reinforcement Learning for Transition-Based Mention Detection","date":"2017-03-13","arxiv_id":"1703.04489","repositories_listed":0,"syntology":null},{"url":null,"slug":"sensor-fusion-for-robot-control-through-deep","title":"Sensor Fusion for Robot Control through Deep Reinforcement Learning","date":"2017-03-13","arxiv_id":"1703.04550","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-a-formal-model-of-cognitive-synergy","title":"Toward a Formal Model of Cognitive Synergy","date":"2017-03-13","arxiv_id":"1703.04361","repositories_listed":0,"syntology":null},{"url":null,"slug":"micro-objective-learning-accelerating-deep","title":"Micro-Objective Learning : Accelerating Deep Reinforcement Learning through the Discovery of Continuous Subgoals","date":"2017-03-11","arxiv_id":"1703.03933","repositories_listed":0,"syntology":null},{"url":null,"slug":"communications-that-emerge-through","title":"Communications that Emerge through Reinforcement Learning Using a (Recurrent) Neural Network","date":"2017-03-10","arxiv_id":"1703.03543","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-feature-selection-for","title":"Sample Efficient Feature Selection for Factored MDPs","date":"2017-03-09","arxiv_id":"1703.03454","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-can-you-do-with-a-rock-affordance","title":"What can you do with a rock? Affordance extraction via word embeddings","date":"2017-03-09","arxiv_id":"1703.03429","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-invariant-feature-spaces-to-transfer","title":"Learning Invariant Feature Spaces to Transfer Skills with Reinforcement Learning","date":"2017-03-08","arxiv_id":"1703.02949","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactics-of-adversarial-attack-on-deep","title":"Tactics of Adversarial Attack on Deep Reinforcement Learning Agents","date":"2017-03-08","arxiv_id":"1703.06748","repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-structured-reinforcement-learning-for","title":"Tree-Structured Reinforcement Learning for Sequential Object Localization","date":"2017-03-08","arxiv_id":"1703.02710","repositories_listed":0,"syntology":null},{"url":null,"slug":"functions-that-emerge-through-end-to-end","title":"Functions that Emerge through End-to-End Reinforcement Learning - The Direction for Artificial General Intelligence -","date":"2017-03-07","arxiv_id":"1703.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"surprise-based-intrinsic-motivation-for-deep","title":"Surprise-Based Intrinsic Motivation for Deep Reinforcement Learning","date":"2017-03-06","arxiv_id":"1703.01732","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-reinforcement-learning-with","title":"Actor-Critic Reinforcement Learning with Simultaneous Human Control and Feedback","date":"2017-03-03","arxiv_id":"1703.01274","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeply-aggrevated-differentiable-imitation","title":"Deeply AggreVaTeD: Differentiable Imitation Learning for Sequential Prediction","date":"2017-03-03","arxiv_id":"1703.01030","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-step-reinforcement-learning-a-unifying","title":"Multi-step Reinforcement Learning: A Unifying Algorithm","date":"2017-03-03","arxiv_id":"1703.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-basis-function-adaptation-for","title":"Unsupervised Basis Function Adaptation for Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01026","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtual-vs-real-trading-off-simulations-and","title":"Virtual vs. Real: Trading Off Simulations and Physical Experiments in Reinforcement Learning with Bayesian Optimization","date":"2017-03-03","arxiv_id":"1703.01250","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-conversational-systems-that","title":"Learning Conversational Systems that Interleave Task and Non-Task Content","date":"2017-03-01","arxiv_id":"1703.00099","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-optimize-neural-nets","title":"Learning to Optimize Neural Nets","date":"2017-03-01","arxiv_id":"1703.00441","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysing-congestion-problems-in-multi-agent","title":"Analysing Congestion Problems in Multi-agent Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08736","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-agent-expertise-in-ms-pac-man","title":"Analysis of Agent Expertise in Ms. Pac-Man using Value-of-Information-based Policies","date":"2017-02-28","arxiv_id":"1702.08628","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-what-data-to-learn","title":"Learning What Data to Learn","date":"2017-02-28","arxiv_id":"1702.08635","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaffolding-networks-incremental-learning-and","title":"Scaffolding Networks: Incremental Learning and Teaching Through Questioning","date":"2017-02-28","arxiv_id":"1702.08653","repositories_listed":0,"syntology":null},{"url":null,"slug":"show-attend-and-interact-perceivable-human","title":"Show, Attend and Interact: Perceivable Human-Robot Social Interaction through Neural Attention Q-Network","date":"2017-02-28","arxiv_id":"1702.08626","repositories_listed":0,"syntology":null},{"url":"/paper/a-dataset-for-developing-and-benchmarking","slug":"a-dataset-for-developing-and-benchmarking","title":"A Dataset for Developing and Benchmarking Active Vision","date":"2017-02-27","arxiv_id":"1702.08272","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-for-air-hockey-striking","title":"Learning Control for Air Hockey Striking using Deep Reinforcement Learning","date":"2017-02-26","arxiv_id":"1702.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-variance-reduction-methods-for","title":"Stochastic Variance Reduction Methods for Policy Evaluation","date":"2017-02-25","arxiv_id":"1702.07944","repositories_listed":0,"syntology":null},{"url":null,"slug":"changing-model-behavior-at-test-time-using","title":"Changing Model Behavior at Test-Time Using Reinforcement Learning","date":"2017-02-24","arxiv_id":"1702.07780","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-gene-regulatory-networks-with","title":"Control of Gene Regulatory Networks with Noisy Measurements and Uncertain Inputs","date":"2017-02-24","arxiv_id":"1702.07652","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-meta-learning-by-parallel-algorithm","title":"Online Meta-learning by Parallel Algorithm Competition","date":"2017-02-24","arxiv_id":"1702.07490","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-gains-social-intelligence-through","title":"Robot gains Social Intelligence through Multimodal Deep Reinforcement Learning","date":"2017-02-24","arxiv_id":"1702.07492","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-goal-based-movement-model-for-continuous","title":"A Goal-Based Movement Model for Continuous Multi-Agent Tasks","date":"2017-02-23","arxiv_id":"1702.07319","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-representation-for-lifetime-value","title":"Automatic Representation for Lifetime Value Recommender Systems","date":"2017-02-23","arxiv_id":"1702.07125","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-distillation-for-controlling-specificity","title":"Data Distillation for Controlling Specificity in Dialogue Generation","date":"2017-02-22","arxiv_id":"1702.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-argument","title":"Reinforcement Learning Based Argument Component Detection","date":"2017-02-21","arxiv_id":"1702.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-repeat-fine-grained-action","title":"Learning to Repeat: Fine Grained Action Repetition for Deep Reinforcement Learning","date":"2017-02-20","arxiv_id":"1702.06054","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-deep-reinforcement-learning-for-1","title":"Collaborative Deep Reinforcement Learning for Joint Object Search","date":"2017-02-18","arxiv_id":"1702.05573","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-robots-to-communicate-their","title":"Enabling Robots to Communicate their Objectives","date":"2017-02-11","arxiv_id":"1702.03465","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-policy-gradient-methods-for-improving","title":"Batch Policy Gradient Methods for Improving Neural Conversation Models","date":"2017-02-10","arxiv_id":"1702.03334","repositories_listed":0,"syntology":null},{"url":"/paper/sigmoid-weighted-linear-units-for-neural","slug":"sigmoid-weighted-linear-units-for-neural","title":"Sigmoid-Weighted Linear Units for Neural Network Function Approximation in Reinforcement Learning","date":"2017-02-10","arxiv_id":"1702.03118","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-qa-with-generative-domain","title":"Semi-Supervised QA with Generative Domain-Adaptive Nets","date":"2017-02-07","arxiv_id":"1702.02206","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentional-network-for-visual-object","title":"Attentional Network for Visual Object Detection","date":"2017-02-06","arxiv_id":"1702.01478","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-reinforcement-learning-for","title":"Uncertainty-Aware Reinforcement Learning for Collision Avoidance","date":"2017-02-03","arxiv_id":"1702.01182","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robotic-1","title":"Deep Reinforcement Learning for Robotic Manipulation-The state of the art","date":"2017-01-31","arxiv_id":"1701.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-visual-object","title":"Deep Reinforcement Learning for Visual Object Tracking in Videos","date":"2017-01-31","arxiv_id":"1701.08936","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-level-control-of-ramp-metering-based","title":"Expert Level control of Ramp Metering based on Multi-task Deep Reinforcement Learning","date":"2017-01-30","arxiv_id":"1701.08832","repositories_listed":0,"syntology":null},{"url":null,"slug":"flow-navigation-by-smart-microswimmers-via","title":"Flow Navigation by Smart Microswimmers via Reinforcement Learning","date":"2017-01-30","arxiv_id":"1701.08848","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-algorithm-selection","title":"Reinforcement Learning Algorithm Selection","date":"2017-01-30","arxiv_id":"1701.08810","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-control-of-thermostatically","title":"Model-Free Control of Thermostatically Controlled Loads Connected to a District Heating Network","date":"2017-01-27","arxiv_id":"1701.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-architecture-for-deep-learning-and-its","title":"FPGA Architecture for Deep Learning and its application to Planetary Robotics","date":"2017-01-26","arxiv_id":"1701.07543","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-intelligence-approaches-to-ucav","title":"Artificial Intelligence Approaches To UCAV Autonomy","date":"2017-01-24","arxiv_id":"1701.07103","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-an-attention-model-in-an-artificial","title":"Learning an attention model in an artificial visual system","date":"2017-01-24","arxiv_id":"1701.07398","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-what-to-look-in-chest-x-rays-with-a","title":"Learning what to look in chest X-rays with a recurrent visual attention model","date":"2017-01-23","arxiv_id":"1701.06452","repositories_listed":0,"syntology":null},{"url":null,"slug":"binary-matrix-guessing-problem","title":"Binary Matrix Guessing Problem","date":"2017-01-22","arxiv_id":"1701.06167","repositories_listed":0,"syntology":null},{"url":null,"slug":"basic-protocols-in-quantum-reinforcement","title":"Basic protocols in quantum reinforcement learning with superconducting circuits","date":"2017-01-18","arxiv_id":"1701.05131","repositories_listed":0,"syntology":null},{"url":null,"slug":"unknowable-manipulators-social-network","title":"Unknowable Manipulators: Social Network Curator Algorithms","date":"2017-01-17","arxiv_id":"1701.04895","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-agnostic-human-in-the-loop","title":"Agent-Agnostic Human-in-the-Loop Reinforcement Learning","date":"2017-01-15","arxiv_id":"1701.04079","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-and-incremental-learning-of-gaussian","title":"Scalable and Incremental Learning of Gaussian Mixture Models","date":"2017-01-14","arxiv_id":"1701.03940","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-embodied-agents","title":"Reinforcement Learning based Embodied Agents Modelling Human Users Through Interaction and Multi-Sensory Perception","date":"2017-01-09","arxiv_id":"1701.02369","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-neural-network-based-machine","title":"A Review of Neural Network Based Machine Learning Approaches for Rotor Angle Stability Control","date":"2017-01-05","arxiv_id":"1701.01214","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-negotiable-reinforcement-learning","title":"Toward negotiable reinforcement learning: shifting priorities in Pareto optimal sequential decision-making","date":"2017-01-05","arxiv_id":"1701.01302","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-deep-neural-networks-optimizing","title":"Dynamic Deep Neural Networks: Optimizing Accuracy-Efficiency Trade-offs by Selective Execution","date":"2017-01-02","arxiv_id":"1701.00299","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-lambda-least-squares-temporal","title":"Adaptive Lambda Least-Squares Temporal Difference Learning","date":"2016-12-30","arxiv_id":"1612.09465","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-neural-networks-through","title":"Understanding Neural Networks through Representation Erasure","date":"2016-12-24","arxiv_id":"1612.08220","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-person-activity-forecasting-with-online","title":"First-Person Activity Forecasting with Online Inverse Reinforcement Learning","date":"2016-12-22","arxiv_id":"1612.07796","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-deterministic-policy-improvement","title":"Non-Deterministic Policy Improvement Stabilizes Approximated Reinforcement Learning","date":"2016-12-22","arxiv_id":"1612.07548","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-function-approximation-error-for-risk","title":"On the function approximation error for risk-sensitive reinforcement learning","date":"2016-12-22","arxiv_id":"1612.07562","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-is-its-own-reward-self-supervision-for","title":"Loss is its own Reward: Self-Supervision for Reinforcement Learning","date":"2016-12-21","arxiv_id":"1612.07307","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-perceptual-rewards-for-imitation","title":"Unsupervised Perceptual Rewards for Imitation Learning","date":"2016-12-20","arxiv_id":"1612.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-deep-reinforcement-learning-1","title":"Sample-efficient Deep Reinforcement Learning for Dialog Control","date":"2016-12-18","arxiv_id":"1612.06000","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-predict-where-to-look-in","title":"Learning to predict where to look in interactive environments using deep recurrent q-learning","date":"2016-12-17","arxiv_id":"1612.05753","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-quantum","title":"Reinforcement Learning Using Quantum Boltzmann Machines","date":"2016-12-17","arxiv_id":"1612.05695","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-successor","title":"Deep Reinforcement Learning with Successor Features for Navigation across Similar Environments","date":"2016-12-16","arxiv_id":"1612.05533","repositories_listed":0,"syntology":null},{"url":null,"slug":"separation-of-concerns-in-reinforcement","title":"Separation of Concerns in Reinforcement Learning","date":"2016-12-15","arxiv_id":"1612.05159","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-deep-reinforcement-learning-for","title":"End-to-End Deep Reinforcement Learning for Lane Keeping Assist","date":"2016-12-13","arxiv_id":"1612.04340","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-human-domain-knowledge-into","title":"Incorporating Human Domain Knowledge into Large Scale Cost Function Learning","date":"2016-12-13","arxiv_id":"1612.04318","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-attention-driven-approach-of-no-reference","title":"An Attention-Driven Approach of No-Reference Image Quality Assessment","date":"2016-12-12","arxiv_id":"1612.03530","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-drive-using-inverse-reinforcement","title":"Learning to Drive using Inverse Reinforcement Learning and Deep Q-Networks","date":"2016-12-12","arxiv_id":"1612.03653","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-for-real-time","title":"Online Reinforcement Learning for Real-Time Exploration in Continuous State and Action Markov Decision Processes","date":"2016-12-12","arxiv_id":"1612.03780","repositories_listed":0,"syntology":null},{"url":null,"slug":"poseagent-budget-constrained-6d-object-pose","title":"PoseAgent: Budget-Constrained 6D Object Pose Estimation via Reinforcement Learning","date":"2016-12-12","arxiv_id":"1612.03779","repositories_listed":0,"syntology":null}],"record_sha256":"ca63073f94ea9537c752ccc6c1c496e5cb6f65af01b932d6aba289218ed1c138","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}