{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/47","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":47,"pages_in_order":152,"rows_per_page":100,"rows":[4601,4700],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/46","next":"/task/reinforcement-learning-1/papers/48","papers":[{"url":"/paper/leave-no-trace-learning-to-reset-for-safe-and","slug":"leave-no-trace-learning-to-reset-for-safe-and","title":"Leave no Trace: Learning to Reset for Safe and Autonomous Reinforcement Learning","date":"2017-11-18","arxiv_id":"1711.06782","repositories_listed":1,"syntology":null},{"url":"/paper/run-skeleton-run-skeletal-model-in-a-physics","slug":"run-skeleton-run-skeletal-model-in-a-physics","title":"Run, skeleton, run: skeletal model in a physics-based simulation","date":"2017-11-18","arxiv_id":"1711.06922","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-policy-gradients","slug":"hindsight-policy-gradients","title":"Hindsight policy gradients","date":"2017-11-16","arxiv_id":"1711.06006","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hindsight-policy-gradients#ran","syntology_url":"https://syntology.ai/paper/1711.06006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.06006"}},"official":null}},{"url":"/paper/classical-structured-prediction-losses-for","slug":"classical-structured-prediction-losses-for","title":"Classical Structured Prediction Losses for Sequence to Sequence Learning","date":"2017-11-14","arxiv_id":"1711.04956","repositories_listed":1,"syntology":null},{"url":"/paper/towards-the-use-of-deep-reinforcement","slug":"towards-the-use-of-deep-reinforcement","title":"Towards the Use of Deep Reinforcement Learning with Global Policy For Query-based Extractive Summarisation","date":"2017-11-10","arxiv_id":"1711.03859","repositories_listed":1,"syntology":null},{"url":"/paper/latentpoison-adversarial-attacks-on-the","slug":"latentpoison-adversarial-attacks-on-the","title":"LatentPoison - Adversarial Attacks On The Latent Space","date":"2017-11-08","arxiv_id":"1711.02879","repositories_listed":1,"syntology":null},{"url":"/paper/can-deep-reinforcement-learning-solve-erdos","slug":"can-deep-reinforcement-learning-solve-erdos","title":"Can Deep Reinforcement Learning Solve Erdos-Selfridge-Spencer Games?","date":"2017-11-07","arxiv_id":"1711.02301","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-game-theoretic-approach-to","slug":"a-unified-game-theoretic-approach-to","title":"A Unified Game-Theoretic Approach to Multiagent Reinforcement Learning","date":"2017-11-02","arxiv_id":"1711.00832","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-coordination-of-working-memory-and","slug":"adaptive-coordination-of-working-memory-and","title":"Adaptive coordination of working-memory and reinforcement learning in non-human primates performing a trial-and-error problem solving task","date":"2017-11-02","arxiv_id":"1711.00698","repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-for-partially-observable","slug":"regret-minimization-for-partially-observable","title":"Regret Minimization for Partially Observable Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11424","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regret-minimization-for-partially-observable#ran","syntology_url":"https://syntology.ai/paper/1710.11424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11424"}},"official":null}},{"url":"/paper/treeqn-and-atreec-differentiable-tree","slug":"treeqn-and-atreec-differentiable-tree","title":"TreeQN and ATreeC: Differentiable Tree-Structured Models for Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11417","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/treeqn-and-atreec-differentiable-tree#ran","syntology_url":"https://syntology.ai/paper/1710.11417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11417"}},"official":{"repos":["oxwhirl/treeqn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/eigenoption-discovery-through-the-deep","slug":"eigenoption-discovery-through-the-deep","title":"Eigenoption Discovery through the Deep Successor Representation","date":"2017-10-30","arxiv_id":"1710.11089","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eigenoption-discovery-through-the-deep#ran","syntology_url":"https://syntology.ai/paper/1710.11089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11089"}},"official":null}},{"url":"/paper/predicting-head-movement-in-panoramic-video-a","slug":"predicting-head-movement-in-panoramic-video-a","title":"Predicting Head Movement in Panoramic Video: A Deep Reinforcement Learning Approach","date":"2017-10-30","arxiv_id":"1710.10755","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-tower-network-a-novel-deep","slug":"generalization-tower-network-a-novel-deep","title":"Generalization Tower Network: A Novel Deep Neural Network Architecture for Multi-Task Learning","date":"2017-10-27","arxiv_id":"1710.10036","repositories_listed":1,"syntology":null},{"url":"/paper/learning-approximate-stochastic-transition","slug":"learning-approximate-stochastic-transition","title":"Learning Approximate Stochastic Transition Models","date":"2017-10-26","arxiv_id":"1710.09718","repositories_listed":1,"syntology":null},{"url":"/paper/decomposition-of-uncertainty-in-bayesian-deep","slug":"decomposition-of-uncertainty-in-bayesian-deep","title":"Decomposition of Uncertainty in Bayesian Deep Learning for Efficient and Risk-sensitive Learning","date":"2017-10-19","arxiv_id":"1710.07283","repositories_listed":1,"syntology":null},{"url":"/paper/the-effects-of-memory-replay-in-reinforcement","slug":"the-effects-of-memory-replay-in-reinforcement","title":"The Effects of Memory Replay in Reinforcement Learning","date":"2017-10-18","arxiv_id":"1710.06574","repositories_listed":1,"syntology":null},{"url":"/paper/learning-complex-dexterous-manipulation-with","slug":"learning-complex-dexterous-manipulation-with","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","date":"2017-09-28","arxiv_id":"1709.10087","repositories_listed":1,"syntology":null},{"url":"/paper/cold-start-reinforcement-learning-with","slug":"cold-start-reinforcement-learning-with","title":"Cold-Start Reinforcement Learning with Softmax Policy Gradient","date":"2017-09-27","arxiv_id":"1709.09346","repositories_listed":1,"syntology":null},{"url":"/paper/mdp-environments-for-the-openai-gym","slug":"mdp-environments-for-the-openai-gym","title":"MDP environments for the OpenAI Gym","date":"2017-09-26","arxiv_id":"1709.09069","repositories_listed":1,"syntology":null},{"url":"/paper/optiongan-learning-joint-reward-policy","slug":"optiongan-learning-joint-reward-policy","title":"OptionGAN: Learning Joint Reward-Policy Options using Generative Adversarial Inverse Reinforcement Learning","date":"2017-09-20","arxiv_id":"1709.06683","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-event-driven","slug":"deep-reinforcement-learning-for-event-driven","title":"Deep Reinforcement Learning for Event-Driven Multi-Agent Decision Processes","date":"2017-09-19","arxiv_id":"1709.06656","repositories_listed":1,"syntology":null},{"url":"/paper/guided-deep-reinforcement-learning-for-swarm","slug":"guided-deep-reinforcement-learning-for-swarm","title":"Guided Deep Reinforcement Learning for Swarm Systems","date":"2017-09-18","arxiv_id":"1709.06011","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-deep-reinforcement-learning-for-swarm#ran","syntology_url":"https://syntology.ai/paper/1709.06011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.06011"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for","slug":"deep-reinforcement-learning-for","title":"Deep Reinforcement Learning for Conversational AI","date":"2017-09-15","arxiv_id":"1709.05067","repositories_listed":1,"syntology":null},{"url":"/paper/shapechanger-environments-for-transfer","slug":"shapechanger-environments-for-transfer","title":"Shapechanger: Environments for Transfer Learning","date":"2017-09-15","arxiv_id":"1709.05070","repositories_listed":1,"syntology":null},{"url":"/paper/automated-cloud-provisioning-on-aws-using","slug":"automated-cloud-provisioning-on-aws-using","title":"Automated Cloud Provisioning on AWS using Deep Reinforcement Learning","date":"2017-09-13","arxiv_id":"1709.04305","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-descent-search-and-its-acceleration","slug":"mirror-descent-search-and-its-acceleration","title":"Mirror Descent Search and its Acceleration","date":"2017-09-08","arxiv_id":"1709.02535","repositories_listed":1,"syntology":null},{"url":"/paper/prosocial-learning-agents-solve-generalized","slug":"prosocial-learning-agents-solve-generalized","title":"Prosocial learning agents solve generalized Stag Hunts better than selfish ones","date":"2017-09-08","arxiv_id":"1709.02865","repositories_listed":1,"syntology":null},{"url":"/paper/speeding-up-reinforcement-learning-based","slug":"speeding-up-reinforcement-learning-based","title":"Speeding up Reinforcement Learning-based Information Extraction Training using Asynchronous Methods","date":"2017-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-via-shielding","slug":"safe-reinforcement-learning-via-shielding","title":"Safe Reinforcement Learning via Shielding","date":"2017-08-29","arxiv_id":"1708.08611","repositories_listed":1,"syntology":null},{"url":"/paper/deep-object-centric-representations-for","slug":"deep-object-centric-representations-for","title":"Deep Object-Centric Representations for Generalizable Robot Learning","date":"2017-08-14","arxiv_id":"1708.04225","repositories_listed":1,"syntology":null},{"url":"/paper/group-driven-reinforcement-learning-for","slug":"group-driven-reinforcement-learning-for","title":"Group-driven Reinforcement Learning for Personalized mHealth Intervention","date":"2017-08-14","arxiv_id":"1708.04001","repositories_listed":1,"syntology":null},{"url":"/paper/reproducibility-of-benchmarked-deep","slug":"reproducibility-of-benchmarked-deep","title":"Reproducibility of Benchmarked Deep Reinforcement Learning Tasks for Continuous Control","date":"2017-08-10","arxiv_id":"1708.04133","repositories_listed":1,"syntology":null},{"url":"/paper/learning-how-to-active-learn-a-deep","slug":"learning-how-to-active-learn-a-deep","title":"Learning how to Active Learn: A Deep Reinforcement Learning Approach","date":"2017-08-08","arxiv_id":"1708.02383","repositories_listed":1,"syntology":null},{"url":"/paper/variational-generative-stochastic-networks","slug":"variational-generative-stochastic-networks","title":"Variational Generative Stochastic Networks with Collaborative Shaping","date":"2017-08-02","arxiv_id":"1708.00805","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-language-for-transfer-in-deep","slug":"grounding-language-for-transfer-in-deep","title":"Grounding Language for Transfer in Deep Reinforcement Learning","date":"2017-08-01","arxiv_id":"1708.00133","repositories_listed":1,"syntology":null},{"url":"/paper/darla-improving-zero-shot-transfer-in","slug":"darla-improving-zero-shot-transfer-in","title":"DARLA: Improving Zero-Shot Transfer in Reinforcement Learning","date":"2017-07-26","arxiv_id":"1707.08475","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-bandit-neural","slug":"reinforcement-learning-for-bandit-neural","title":"Reinforcement Learning for Bandit Neural Machine Translation with Simulated Human Feedback","date":"2017-07-24","arxiv_id":"1707.07402","repositories_listed":1,"syntology":null},{"url":"/paper/trial-without-error-towards-safe","slug":"trial-without-error-towards-safe","title":"Trial without Error: Towards Safe Reinforcement Learning via Human Intervention","date":"2017-07-17","arxiv_id":"1707.05173","repositories_listed":1,"syntology":null},{"url":"/paper/lenient-multi-agent-deep-reinforcement","slug":"lenient-multi-agent-deep-reinforcement","title":"Lenient Multi-Agent Deep Reinforcement Learning","date":"2017-07-14","arxiv_id":"1707.04402","repositories_listed":1,"syntology":null},{"url":"/paper/representation-learning-for-grounded-spatial","slug":"representation-learning-for-grounded-spatial","title":"Representation Learning for Grounded Spatial Reasoning","date":"2017-07-13","arxiv_id":"1707.03938","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-behaviors-from-motion-capture","slug":"learning-human-behaviors-from-motion-capture","title":"Learning human behaviors from motion capture by adversarial imitation","date":"2017-07-07","arxiv_id":"1707.02201","repositories_listed":1,"syntology":null},{"url":"/paper/trust-pcl-an-off-policy-trust-region-method","slug":"trust-pcl-an-off-policy-trust-region-method","title":"Trust-PCL: An Off-Policy Trust Region Method for Continuous Control","date":"2017-07-06","arxiv_id":"1707.01891","repositories_listed":1,"syntology":null},{"url":"/paper/maintaining-cooperation-in-complex-social","slug":"maintaining-cooperation-in-complex-social","title":"Maintaining cooperation in complex social dilemmas using deep reinforcement learning","date":"2017-07-04","arxiv_id":"1707.01068","repositories_listed":1,"syntology":null},{"url":"/paper/action-decision-networks-for-visual-tracking","slug":"action-decision-networks-for-visual-tracking","title":"Action-Decision Networks for Visual Tracking With Deep Reinforcement Learning","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-sequence-model-training-via-divergence","slug":"neural-sequence-model-training-via-divergence","title":"Neural Sequence Model Training via $α$-divergence Minimization","date":"2017-06-30","arxiv_id":"1706.10031","repositories_listed":1,"syntology":null},{"url":"/paper/neural-slam-learning-to-explore-with-external","slug":"neural-slam-learning-to-explore-with-external","title":"Neural SLAM: Learning to Explore with External Memory","date":"2017-06-29","arxiv_id":"1706.09520","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-in-feature-space-for","slug":"count-based-exploration-in-feature-space-for","title":"Count-Based Exploration in Feature Space for Reinforcement Learning","date":"2017-06-25","arxiv_id":"1706.08090","repositories_listed":1,"syntology":null},{"url":"/paper/a-self-adaptive-proposal-model-for-temporal","slug":"a-self-adaptive-proposal-model-for-temporal","title":"A Self-Adaptive Proposal Model for Temporal Action Detection based on Reinforcement Learning","date":"2017-06-22","arxiv_id":"1706.07251","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-reinforcement-learning-with","slug":"data-efficient-reinforcement-learning-with","title":"Data-Efficient Reinforcement Learning with Probabilistic Model Predictive Control","date":"2017-06-20","arxiv_id":"1706.06491","repositories_listed":1,"syntology":null},{"url":"/paper/dex-incremental-learning-for-complex","slug":"dex-incremental-learning-for-complex","title":"Dex: Incremental Learning for Complex Environments in Deep Reinforcement Learning","date":"2017-06-19","arxiv_id":"1706.05749","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-task-generalization-with-multi-task","slug":"zero-shot-task-generalization-with-multi-task","title":"Zero-Shot Task Generalization with Multi-Task Deep Reinforcement Learning","date":"2017-06-15","arxiv_id":"1706.05064","repositories_listed":1,"syntology":null},{"url":"/paper/device-placement-optimization-with","slug":"device-placement-optimization-with","title":"Device Placement Optimization with Reinforcement Learning","date":"2017-06-13","arxiv_id":"1706.04972","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-reward-architecture-for-reinforcement","slug":"hybrid-reward-architecture-for-reinforcement","title":"Hybrid Reward Architecture for Reinforcement Learning","date":"2017-06-13","arxiv_id":"1706.04208","repositories_listed":1,"syntology":null},{"url":"/paper/objective-reinforced-generative-adversarial","slug":"objective-reinforced-generative-adversarial","title":"Objective-Reinforced Generative Adversarial Networks (ORGAN) for Sequence Generation Models","date":"2017-05-30","arxiv_id":"1705.10843","repositories_listed":1,"syntology":null},{"url":"/paper/universal-reinforcement-learning-algorithms","slug":"universal-reinforcement-learning-algorithms","title":"Universal Reinforcement Learning Algorithms: Survey and Experiments","date":"2017-05-30","arxiv_id":"1705.10557","repositories_listed":1,"syntology":null},{"url":"/paper/free-energy-based-reinforcement-learning","slug":"free-energy-based-reinforcement-learning","title":"Free energy-based reinforcement learning using a quantum processor","date":"2017-05-29","arxiv_id":"1706.00074","repositories_listed":1,"syntology":null},{"url":"/paper/latent-intention-dialogue-models","slug":"latent-intention-dialogue-models","title":"Latent Intention Dialogue Models","date":"2017-05-29","arxiv_id":"1705.10229","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-corrupted","slug":"reinforcement-learning-with-a-corrupted","title":"Reinforcement Learning with a Corrupted Reward Channel","date":"2017-05-23","arxiv_id":"1705.08417","repositories_listed":1,"syntology":null},{"url":"/paper/safe-model-based-reinforcement-learning-with","slug":"safe-model-based-reinforcement-learning-with","title":"Safe Model-based Reinforcement Learning with Stability Guarantees","date":"2017-05-23","arxiv_id":"1705.08551","repositories_listed":1,"syntology":null},{"url":"/paper/aixijs-a-software-demo-for-general","slug":"aixijs-a-software-demo-for-general","title":"AIXIjs: A Software Demo for General Reinforcement Learning","date":"2017-05-22","arxiv_id":"1705.07615","repositories_listed":1,"syntology":null},{"url":"/paper/guide-actor-critic-for-continuous-control","slug":"guide-actor-critic-for-continuous-control","title":"Guide Actor-Critic for Continuous Control","date":"2017-05-22","arxiv_id":"1705.07606","repositories_listed":1,"syntology":null},{"url":"/paper/feature-control-as-intrinsic-motivation-for","slug":"feature-control-as-intrinsic-motivation-for","title":"Feature Control as Intrinsic Motivation for Hierarchical Reinforcement Learning","date":"2017-05-18","arxiv_id":"1705.06769","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-goal-generation-for-reinforcement","slug":"automatic-goal-generation-for-reinforcement","title":"Automatic Goal Generation for Reinforcement Learning Agents","date":"2017-05-17","arxiv_id":"1705.06366","repositories_listed":1,"syntology":null},{"url":"/paper/integral-policy-iterations-for-reinforcement","slug":"integral-policy-iterations-for-reinforcement","title":"Policy Iterations for Reinforcement Learning Problems in Continuous Time and Space -- Fundamental Theory and Methods","date":"2017-05-09","arxiv_id":"1705.03520","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multimodal-transition-dynamics-for","slug":"learning-multimodal-transition-dynamics-for","title":"Learning Multimodal Transition Dynamics for Model-Based Reinforcement Learning","date":"2017-05-01","arxiv_id":"1705.00470","repositories_listed":1,"syntology":null},{"url":"/paper/mapping-instructions-and-visual-observations","slug":"mapping-instructions-and-visual-observations","title":"Mapping Instructions and Visual Observations to Actions with Reinforcement Learning","date":"2017-04-28","arxiv_id":"1704.08795","repositories_listed":1,"syntology":null},{"url":"/paper/on-improving-deep-reinforcement-learning-for","slug":"on-improving-deep-reinforcement-learning-for","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2017-04-26","arxiv_id":"1704.07978","repositories_listed":1,"syntology":null},{"url":"/paper/modular-multi-objective-deep-reinforcement","slug":"modular-multi-objective-deep-reinforcement","title":"Modular Multi-Objective Deep Reinforcement Learning with Decision Values","date":"2017-04-21","arxiv_id":"1704.06676","repositories_listed":1,"syntology":null},{"url":"/paper/beating-atari-with-natural-language-guided","slug":"beating-atari-with-natural-language-guided","title":"Beating Atari with Natural Language Guided Reinforcement Learning","date":"2017-04-18","arxiv_id":"1704.05539","repositories_listed":1,"syntology":null},{"url":"/paper/muse-modularizing-unsupervised-sense","slug":"muse-modularizing-unsupervised-sense","title":"MUSE: Modularizing Unsupervised Sense Embeddings","date":"2017-04-15","arxiv_id":"1704.04601","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-differentiable-relaxations-of","slug":"optimizing-differentiable-relaxations-of","title":"Optimizing Differentiable Relaxations of Coreference Evaluation Metrics","date":"2017-04-14","arxiv_id":"1704.04451","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-framework-for","slug":"deep-reinforcement-learning-framework-for","title":"Deep Reinforcement Learning framework for Autonomous Driving","date":"2017-04-08","arxiv_id":"1704.02532","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-simplification-with-deep","slug":"sentence-simplification-with-deep","title":"Sentence Simplification with Deep Reinforcement Learning","date":"2017-03-31","arxiv_id":"1703.10931","repositories_listed":1,"syntology":null},{"url":"/paper/faster-reinforcement-learning-using-active","slug":"faster-reinforcement-learning-using-active","title":"Faster Reinforcement Learning Using Active Simulators","date":"2017-03-22","arxiv_id":"1703.07853","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-pac-and-regret-uniform-pac-bounds","slug":"unifying-pac-and-regret-uniform-pac-bounds","title":"Unifying PAC and Regret: Uniform PAC Bounds for Episodic Reinforcement Learning","date":"2017-03-22","arxiv_id":"1703.07710","repositories_listed":1,"syntology":null},{"url":"/paper/black-box-data-efficient-policy-search-for","slug":"black-box-data-efficient-policy-search-for","title":"Black-Box Data-efficient Policy Search for Robotics","date":"2017-03-21","arxiv_id":"1703.07261","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-regret-bounds-for-reinforcement","slug":"minimax-regret-bounds-for-reinforcement","title":"Minimax Regret Bounds for Reinforcement Learning","date":"2017-03-16","arxiv_id":"1703.05449","repositories_listed":1,"syntology":null},{"url":"/paper/deep-variation-structured-reinforcement","slug":"deep-variation-structured-reinforcement","title":"Deep Variation-structured Reinforcement Learning for Visual Relationship and Attribute Detection","date":"2017-03-08","arxiv_id":"1703.03054","repositories_listed":1,"syntology":null},{"url":"/paper/third-person-imitation-learning","slug":"third-person-imitation-learning","title":"Third-Person Imitation Learning","date":"2017-03-06","arxiv_id":"1703.01703","repositories_listed":1,"syntology":null},{"url":"/paper/ex2-exploration-with-exemplar-models-for-deep","slug":"ex2-exploration-with-exemplar-models-for-deep","title":"EX2: Exploration with Exemplar Models for Deep Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01260","repositories_listed":1,"syntology":null},{"url":"/paper/feudal-networks-for-hierarchical","slug":"feudal-networks-for-hierarchical","title":"FeUdal Networks for Hierarchical Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01161","repositories_listed":1,"syntology":null},{"url":"/paper/generalised-discount-functions-applied-to-a","slug":"generalised-discount-functions-applied-to-a","title":"Generalised Discount Functions applied to a Monte-Carlo AImu Implementation","date":"2017-03-03","arxiv_id":"1703.01358","repositories_listed":1,"syntology":null},{"url":"/paper/a-laplacian-framework-for-option-discovery-in","slug":"a-laplacian-framework-for-option-discovery-in","title":"A Laplacian Framework for Option Discovery in Reinforcement Learning","date":"2017-03-02","arxiv_id":"1703.00956","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-pivoting-task","slug":"reinforcement-learning-for-pivoting-task","title":"Reinforcement Learning for Pivoting Task","date":"2017-03-01","arxiv_id":"1703.00472","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-value-and-policy","slug":"bridging-the-gap-between-value-and-policy","title":"Bridging the Gap Between Value and Policy Based Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08892","repositories_listed":1,"syntology":null},{"url":"/paper/neural-map-structured-memory-for-deep","slug":"neural-map-structured-memory-for-deep","title":"Neural Map: Structured Memory for Deep Reinforcement Learning","date":"2017-02-27","arxiv_id":"1702.08360","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-error-propagation-through","slug":"tackling-error-propagation-through","title":"Tackling Error Propagation through Reinforcement Learning: A Case of Greedy Dependency Parsing","date":"2017-02-22","arxiv_id":"1702.06794","repositories_listed":1,"syntology":null},{"url":"/paper/beating-the-worlds-best-at-super-smash-bros","slug":"beating-the-worlds-best-at-super-smash-bros","title":"Beating the World's Best at Super Smash Bros. with Deep Reinforcement Learning","date":"2017-02-21","arxiv_id":"1702.06230","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-visual-tracking-by-deep-reinforced","slug":"real-time-visual-tracking-by-deep-reinforced","title":"Real-time visual tracking by deep reinforced decision making","date":"2017-02-21","arxiv_id":"1702.06291","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-common-implementation-of","slug":"towards-a-common-implementation-of","title":"Towards a Common Implementation of Reinforcement Learning for Multiple Robotic Tasks","date":"2017-02-21","arxiv_id":"1702.06329","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-deep-reinforcement-learning","slug":"collaborative-deep-reinforcement-learning","title":"Collaborative Deep Reinforcement Learning","date":"2017-02-19","arxiv_id":"1702.05796","repositories_listed":1,"syntology":null},{"url":"/paper/pathnet-evolution-channels-gradient-descent","slug":"pathnet-evolution-channels-gradient-descent","title":"PathNet: Evolution Channels Gradient Descent in Super Neural Networks","date":"2017-01-30","arxiv_id":"1701.08734","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pathnet-evolution-channels-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/1701.08734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.08734"}},"official":null}},{"url":"/paper/vulnerability-of-deep-reinforcement-learning","slug":"vulnerability-of-deep-reinforcement-learning","title":"Vulnerability of Deep Reinforcement Learning to Policy Induction Attacks","date":"2017-01-16","arxiv_id":"1701.04143","repositories_listed":1,"syntology":null},{"url":"/paper/near-optimal-behavior-via-approximate-state","slug":"near-optimal-behavior-via-approximate-state","title":"Near Optimal Behavior via Approximate State Abstraction","date":"2017-01-15","arxiv_id":"1701.04113","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-bidding-by-reinforcement-learning","slug":"real-time-bidding-by-reinforcement-learning","title":"Real-Time Bidding by Reinforcement Learning in Display Advertising","date":"2017-01-10","arxiv_id":"1701.02490","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-via-recurrent","slug":"reinforcement-learning-via-recurrent","title":"Reinforcement Learning via Recurrent Convolutional Neural Networks","date":"2017-01-09","arxiv_id":"1701.02392","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-deep-network-solutions-for","slug":"a-survey-of-deep-network-solutions-for","title":"A Survey of Deep Network Solutions for Learning Control in Robotics: From Reinforcement to Imitation","date":"2016-12-21","arxiv_id":"1612.07139","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-models-for-model-based","slug":"self-correcting-models-for-model-based","title":"Self-Correcting Models for Model-Based Reinforcement Learning","date":"2016-12-19","arxiv_id":"1612.06018","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-optimization-with-robust-bayesian","slug":"bayesian-optimization-with-robust-bayesian","title":"Bayesian Optimization with Robust Bayesian Neural Networks","date":"2016-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"77b003aa92ac0d0d8422265632ceccb8146174198802a7dc50f62b5b770b077a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}