{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/41","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":41,"pages_in_order":135,"rows_per_page":100,"rows":[4001,4100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/40","next":"/task/reinforcement-learning-2/papers/42","papers":[{"url":"/paper/decomposition-of-uncertainty-in-bayesian-deep","slug":"decomposition-of-uncertainty-in-bayesian-deep","title":"Decomposition of Uncertainty in Bayesian Deep Learning for Efficient and Risk-sensitive Learning","date":"2017-10-19","arxiv_id":"1710.07283","repositories_listed":1,"syntology":null},{"url":"/paper/the-effects-of-memory-replay-in-reinforcement","slug":"the-effects-of-memory-replay-in-reinforcement","title":"The Effects of Memory Replay in Reinforcement Learning","date":"2017-10-18","arxiv_id":"1710.06574","repositories_listed":1,"syntology":null},{"url":"/paper/learning-complex-dexterous-manipulation-with","slug":"learning-complex-dexterous-manipulation-with","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","date":"2017-09-28","arxiv_id":"1709.10087","repositories_listed":1,"syntology":null},{"url":"/paper/cold-start-reinforcement-learning-with","slug":"cold-start-reinforcement-learning-with","title":"Cold-Start Reinforcement Learning with Softmax Policy Gradient","date":"2017-09-27","arxiv_id":"1709.09346","repositories_listed":1,"syntology":null},{"url":"/paper/mdp-environments-for-the-openai-gym","slug":"mdp-environments-for-the-openai-gym","title":"MDP environments for the OpenAI Gym","date":"2017-09-26","arxiv_id":"1709.09069","repositories_listed":1,"syntology":null},{"url":"/paper/optiongan-learning-joint-reward-policy","slug":"optiongan-learning-joint-reward-policy","title":"OptionGAN: Learning Joint Reward-Policy Options using Generative Adversarial Inverse Reinforcement Learning","date":"2017-09-20","arxiv_id":"1709.06683","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-event-driven","slug":"deep-reinforcement-learning-for-event-driven","title":"Deep Reinforcement Learning for Event-Driven Multi-Agent Decision Processes","date":"2017-09-19","arxiv_id":"1709.06656","repositories_listed":1,"syntology":null},{"url":"/paper/guided-deep-reinforcement-learning-for-swarm","slug":"guided-deep-reinforcement-learning-for-swarm","title":"Guided Deep Reinforcement Learning for Swarm Systems","date":"2017-09-18","arxiv_id":"1709.06011","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-deep-reinforcement-learning-for-swarm#ran","syntology_url":"https://syntology.ai/paper/1709.06011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.06011"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for","slug":"deep-reinforcement-learning-for","title":"Deep Reinforcement Learning for Conversational AI","date":"2017-09-15","arxiv_id":"1709.05067","repositories_listed":1,"syntology":null},{"url":"/paper/shapechanger-environments-for-transfer","slug":"shapechanger-environments-for-transfer","title":"Shapechanger: Environments for Transfer Learning","date":"2017-09-15","arxiv_id":"1709.05070","repositories_listed":1,"syntology":null},{"url":"/paper/automated-cloud-provisioning-on-aws-using","slug":"automated-cloud-provisioning-on-aws-using","title":"Automated Cloud Provisioning on AWS using Deep Reinforcement Learning","date":"2017-09-13","arxiv_id":"1709.04305","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-descent-search-and-its-acceleration","slug":"mirror-descent-search-and-its-acceleration","title":"Mirror Descent Search and its Acceleration","date":"2017-09-08","arxiv_id":"1709.02535","repositories_listed":1,"syntology":null},{"url":"/paper/prosocial-learning-agents-solve-generalized","slug":"prosocial-learning-agents-solve-generalized","title":"Prosocial learning agents solve generalized Stag Hunts better than selfish ones","date":"2017-09-08","arxiv_id":"1709.02865","repositories_listed":1,"syntology":null},{"url":"/paper/speeding-up-reinforcement-learning-based","slug":"speeding-up-reinforcement-learning-based","title":"Speeding up Reinforcement Learning-based Information Extraction Training using Asynchronous Methods","date":"2017-09-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/safe-reinforcement-learning-via-shielding","slug":"safe-reinforcement-learning-via-shielding","title":"Safe Reinforcement Learning via Shielding","date":"2017-08-29","arxiv_id":"1708.08611","repositories_listed":1,"syntology":null},{"url":"/paper/group-driven-reinforcement-learning-for","slug":"group-driven-reinforcement-learning-for","title":"Group-driven Reinforcement Learning for Personalized mHealth Intervention","date":"2017-08-14","arxiv_id":"1708.04001","repositories_listed":1,"syntology":null},{"url":"/paper/reproducibility-of-benchmarked-deep","slug":"reproducibility-of-benchmarked-deep","title":"Reproducibility of Benchmarked Deep Reinforcement Learning Tasks for Continuous Control","date":"2017-08-10","arxiv_id":"1708.04133","repositories_listed":1,"syntology":null},{"url":"/paper/learning-how-to-active-learn-a-deep","slug":"learning-how-to-active-learn-a-deep","title":"Learning how to Active Learn: A Deep Reinforcement Learning Approach","date":"2017-08-08","arxiv_id":"1708.02383","repositories_listed":1,"syntology":null},{"url":"/paper/variational-generative-stochastic-networks","slug":"variational-generative-stochastic-networks","title":"Variational Generative Stochastic Networks with Collaborative Shaping","date":"2017-08-02","arxiv_id":"1708.00805","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-language-for-transfer-in-deep","slug":"grounding-language-for-transfer-in-deep","title":"Grounding Language for Transfer in Deep Reinforcement Learning","date":"2017-08-01","arxiv_id":"1708.00133","repositories_listed":1,"syntology":null},{"url":"/paper/darla-improving-zero-shot-transfer-in","slug":"darla-improving-zero-shot-transfer-in","title":"DARLA: Improving Zero-Shot Transfer in Reinforcement Learning","date":"2017-07-26","arxiv_id":"1707.08475","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-bandit-neural","slug":"reinforcement-learning-for-bandit-neural","title":"Reinforcement Learning for Bandit Neural Machine Translation with Simulated Human Feedback","date":"2017-07-24","arxiv_id":"1707.07402","repositories_listed":1,"syntology":null},{"url":"/paper/trial-without-error-towards-safe","slug":"trial-without-error-towards-safe","title":"Trial without Error: Towards Safe Reinforcement Learning via Human Intervention","date":"2017-07-17","arxiv_id":"1707.05173","repositories_listed":1,"syntology":null},{"url":"/paper/lenient-multi-agent-deep-reinforcement","slug":"lenient-multi-agent-deep-reinforcement","title":"Lenient Multi-Agent Deep Reinforcement Learning","date":"2017-07-14","arxiv_id":"1707.04402","repositories_listed":1,"syntology":null},{"url":"/paper/representation-learning-for-grounded-spatial","slug":"representation-learning-for-grounded-spatial","title":"Representation Learning for Grounded Spatial Reasoning","date":"2017-07-13","arxiv_id":"1707.03938","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-behaviors-from-motion-capture","slug":"learning-human-behaviors-from-motion-capture","title":"Learning human behaviors from motion capture by adversarial imitation","date":"2017-07-07","arxiv_id":"1707.02201","repositories_listed":1,"syntology":null},{"url":"/paper/maintaining-cooperation-in-complex-social","slug":"maintaining-cooperation-in-complex-social","title":"Maintaining cooperation in complex social dilemmas using deep reinforcement learning","date":"2017-07-04","arxiv_id":"1707.01068","repositories_listed":1,"syntology":null},{"url":"/paper/action-decision-networks-for-visual-tracking","slug":"action-decision-networks-for-visual-tracking","title":"Action-Decision Networks for Visual Tracking With Deep Reinforcement Learning","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-sequence-model-training-via-divergence","slug":"neural-sequence-model-training-via-divergence","title":"Neural Sequence Model Training via $α$-divergence Minimization","date":"2017-06-30","arxiv_id":"1706.10031","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-in-feature-space-for","slug":"count-based-exploration-in-feature-space-for","title":"Count-Based Exploration in Feature Space for Reinforcement Learning","date":"2017-06-25","arxiv_id":"1706.08090","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-reinforcement-learning-with","slug":"data-efficient-reinforcement-learning-with","title":"Data-Efficient Reinforcement Learning with Probabilistic Model Predictive Control","date":"2017-06-20","arxiv_id":"1706.06491","repositories_listed":1,"syntology":null},{"url":"/paper/dex-incremental-learning-for-complex","slug":"dex-incremental-learning-for-complex","title":"Dex: Incremental Learning for Complex Environments in Deep Reinforcement Learning","date":"2017-06-19","arxiv_id":"1706.05749","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-task-generalization-with-multi-task","slug":"zero-shot-task-generalization-with-multi-task","title":"Zero-Shot Task Generalization with Multi-Task Deep Reinforcement Learning","date":"2017-06-15","arxiv_id":"1706.05064","repositories_listed":1,"syntology":null},{"url":"/paper/device-placement-optimization-with","slug":"device-placement-optimization-with","title":"Device Placement Optimization with Reinforcement Learning","date":"2017-06-13","arxiv_id":"1706.04972","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-reward-architecture-for-reinforcement","slug":"hybrid-reward-architecture-for-reinforcement","title":"Hybrid Reward Architecture for Reinforcement Learning","date":"2017-06-13","arxiv_id":"1706.04208","repositories_listed":1,"syntology":null},{"url":"/paper/universal-reinforcement-learning-algorithms","slug":"universal-reinforcement-learning-algorithms","title":"Universal Reinforcement Learning Algorithms: Survey and Experiments","date":"2017-05-30","arxiv_id":"1705.10557","repositories_listed":1,"syntology":null},{"url":"/paper/free-energy-based-reinforcement-learning","slug":"free-energy-based-reinforcement-learning","title":"Free energy-based reinforcement learning using a quantum processor","date":"2017-05-29","arxiv_id":"1706.00074","repositories_listed":1,"syntology":null},{"url":"/paper/latent-intention-dialogue-models","slug":"latent-intention-dialogue-models","title":"Latent Intention Dialogue Models","date":"2017-05-29","arxiv_id":"1705.10229","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-corrupted","slug":"reinforcement-learning-with-a-corrupted","title":"Reinforcement Learning with a Corrupted Reward Channel","date":"2017-05-23","arxiv_id":"1705.08417","repositories_listed":1,"syntology":null},{"url":"/paper/safe-model-based-reinforcement-learning-with","slug":"safe-model-based-reinforcement-learning-with","title":"Safe Model-based Reinforcement Learning with Stability Guarantees","date":"2017-05-23","arxiv_id":"1705.08551","repositories_listed":1,"syntology":null},{"url":"/paper/aixijs-a-software-demo-for-general","slug":"aixijs-a-software-demo-for-general","title":"AIXIjs: A Software Demo for General Reinforcement Learning","date":"2017-05-22","arxiv_id":"1705.07615","repositories_listed":1,"syntology":null},{"url":"/paper/guide-actor-critic-for-continuous-control","slug":"guide-actor-critic-for-continuous-control","title":"Guide Actor-Critic for Continuous Control","date":"2017-05-22","arxiv_id":"1705.07606","repositories_listed":1,"syntology":null},{"url":"/paper/feature-control-as-intrinsic-motivation-for","slug":"feature-control-as-intrinsic-motivation-for","title":"Feature Control as Intrinsic Motivation for Hierarchical Reinforcement Learning","date":"2017-05-18","arxiv_id":"1705.06769","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-goal-generation-for-reinforcement","slug":"automatic-goal-generation-for-reinforcement","title":"Automatic Goal Generation for Reinforcement Learning Agents","date":"2017-05-17","arxiv_id":"1705.06366","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multimodal-transition-dynamics-for","slug":"learning-multimodal-transition-dynamics-for","title":"Learning Multimodal Transition Dynamics for Model-Based Reinforcement Learning","date":"2017-05-01","arxiv_id":"1705.00470","repositories_listed":1,"syntology":null},{"url":"/paper/mapping-instructions-and-visual-observations","slug":"mapping-instructions-and-visual-observations","title":"Mapping Instructions and Visual Observations to Actions with Reinforcement Learning","date":"2017-04-28","arxiv_id":"1704.08795","repositories_listed":1,"syntology":null},{"url":"/paper/on-improving-deep-reinforcement-learning-for","slug":"on-improving-deep-reinforcement-learning-for","title":"On Improving Deep Reinforcement Learning for POMDPs","date":"2017-04-26","arxiv_id":"1704.07978","repositories_listed":1,"syntology":null},{"url":"/paper/modular-multi-objective-deep-reinforcement","slug":"modular-multi-objective-deep-reinforcement","title":"Modular Multi-Objective Deep Reinforcement Learning with Decision Values","date":"2017-04-21","arxiv_id":"1704.06676","repositories_listed":1,"syntology":null},{"url":"/paper/beating-atari-with-natural-language-guided","slug":"beating-atari-with-natural-language-guided","title":"Beating Atari with Natural Language Guided Reinforcement Learning","date":"2017-04-18","arxiv_id":"1704.05539","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-differentiable-relaxations-of","slug":"optimizing-differentiable-relaxations-of","title":"Optimizing Differentiable Relaxations of Coreference Evaluation Metrics","date":"2017-04-14","arxiv_id":"1704.04451","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-framework-for","slug":"deep-reinforcement-learning-framework-for","title":"Deep Reinforcement Learning framework for Autonomous Driving","date":"2017-04-08","arxiv_id":"1704.02532","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-simplification-with-deep","slug":"sentence-simplification-with-deep","title":"Sentence Simplification with Deep Reinforcement Learning","date":"2017-03-31","arxiv_id":"1703.10931","repositories_listed":1,"syntology":null},{"url":"/paper/faster-reinforcement-learning-using-active","slug":"faster-reinforcement-learning-using-active","title":"Faster Reinforcement Learning Using Active Simulators","date":"2017-03-22","arxiv_id":"1703.07853","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-pac-and-regret-uniform-pac-bounds","slug":"unifying-pac-and-regret-uniform-pac-bounds","title":"Unifying PAC and Regret: Uniform PAC Bounds for Episodic Reinforcement Learning","date":"2017-03-22","arxiv_id":"1703.07710","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-regret-bounds-for-reinforcement","slug":"minimax-regret-bounds-for-reinforcement","title":"Minimax Regret Bounds for Reinforcement Learning","date":"2017-03-16","arxiv_id":"1703.05449","repositories_listed":1,"syntology":null},{"url":"/paper/deep-variation-structured-reinforcement","slug":"deep-variation-structured-reinforcement","title":"Deep Variation-structured Reinforcement Learning for Visual Relationship and Attribute Detection","date":"2017-03-08","arxiv_id":"1703.03054","repositories_listed":1,"syntology":null},{"url":"/paper/third-person-imitation-learning","slug":"third-person-imitation-learning","title":"Third-Person Imitation Learning","date":"2017-03-06","arxiv_id":"1703.01703","repositories_listed":1,"syntology":null},{"url":"/paper/ex2-exploration-with-exemplar-models-for-deep","slug":"ex2-exploration-with-exemplar-models-for-deep","title":"EX2: Exploration with Exemplar Models for Deep Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01260","repositories_listed":1,"syntology":null},{"url":"/paper/feudal-networks-for-hierarchical","slug":"feudal-networks-for-hierarchical","title":"FeUdal Networks for Hierarchical Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01161","repositories_listed":1,"syntology":null},{"url":"/paper/generalised-discount-functions-applied-to-a","slug":"generalised-discount-functions-applied-to-a","title":"Generalised Discount Functions applied to a Monte-Carlo AImu Implementation","date":"2017-03-03","arxiv_id":"1703.01358","repositories_listed":1,"syntology":null},{"url":"/paper/a-laplacian-framework-for-option-discovery-in","slug":"a-laplacian-framework-for-option-discovery-in","title":"A Laplacian Framework for Option Discovery in Reinforcement Learning","date":"2017-03-02","arxiv_id":"1703.00956","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-pivoting-task","slug":"reinforcement-learning-for-pivoting-task","title":"Reinforcement Learning for Pivoting Task","date":"2017-03-01","arxiv_id":"1703.00472","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-value-and-policy","slug":"bridging-the-gap-between-value-and-policy","title":"Bridging the Gap Between Value and Policy Based Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08892","repositories_listed":1,"syntology":null},{"url":"/paper/neural-map-structured-memory-for-deep","slug":"neural-map-structured-memory-for-deep","title":"Neural Map: Structured Memory for Deep Reinforcement Learning","date":"2017-02-27","arxiv_id":"1702.08360","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-error-propagation-through","slug":"tackling-error-propagation-through","title":"Tackling Error Propagation through Reinforcement Learning: A Case of Greedy Dependency Parsing","date":"2017-02-22","arxiv_id":"1702.06794","repositories_listed":1,"syntology":null},{"url":"/paper/beating-the-worlds-best-at-super-smash-bros","slug":"beating-the-worlds-best-at-super-smash-bros","title":"Beating the World's Best at Super Smash Bros. with Deep Reinforcement Learning","date":"2017-02-21","arxiv_id":"1702.06230","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-visual-tracking-by-deep-reinforced","slug":"real-time-visual-tracking-by-deep-reinforced","title":"Real-time visual tracking by deep reinforced decision making","date":"2017-02-21","arxiv_id":"1702.06291","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-common-implementation-of","slug":"towards-a-common-implementation-of","title":"Towards a Common Implementation of Reinforcement Learning for Multiple Robotic Tasks","date":"2017-02-21","arxiv_id":"1702.06329","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-deep-reinforcement-learning","slug":"collaborative-deep-reinforcement-learning","title":"Collaborative Deep Reinforcement Learning","date":"2017-02-19","arxiv_id":"1702.05796","repositories_listed":1,"syntology":null},{"url":"/paper/pathnet-evolution-channels-gradient-descent","slug":"pathnet-evolution-channels-gradient-descent","title":"PathNet: Evolution Channels Gradient Descent in Super Neural Networks","date":"2017-01-30","arxiv_id":"1701.08734","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pathnet-evolution-channels-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/1701.08734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.08734"}},"official":null}},{"url":"/paper/vulnerability-of-deep-reinforcement-learning","slug":"vulnerability-of-deep-reinforcement-learning","title":"Vulnerability of Deep Reinforcement Learning to Policy Induction Attacks","date":"2017-01-16","arxiv_id":"1701.04143","repositories_listed":1,"syntology":null},{"url":"/paper/near-optimal-behavior-via-approximate-state","slug":"near-optimal-behavior-via-approximate-state","title":"Near Optimal Behavior via Approximate State Abstraction","date":"2017-01-15","arxiv_id":"1701.04113","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-bidding-by-reinforcement-learning","slug":"real-time-bidding-by-reinforcement-learning","title":"Real-Time Bidding by Reinforcement Learning in Display Advertising","date":"2017-01-10","arxiv_id":"1701.02490","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-via-recurrent","slug":"reinforcement-learning-via-recurrent","title":"Reinforcement Learning via Recurrent Convolutional Neural Networks","date":"2017-01-09","arxiv_id":"1701.02392","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-deep-network-solutions-for","slug":"a-survey-of-deep-network-solutions-for","title":"A Survey of Deep Network Solutions for Learning Control in Robotics: From Reinforcement to Imitation","date":"2016-12-21","arxiv_id":"1612.07139","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-models-for-model-based","slug":"self-correcting-models-for-model-based","title":"Self-Correcting Models for Model-Based Reinforcement Learning","date":"2016-12-19","arxiv_id":"1612.06018","repositories_listed":1,"syntology":null},{"url":"/paper/playing-doom-with-slam-augmented-deep","slug":"playing-doom-with-slam-augmented-deep","title":"Playing Doom with SLAM-Augmented Deep Reinforcement Learning","date":"2016-12-01","arxiv_id":"1612.00380","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-domain","slug":"deep-reinforcement-learning-for-multi-domain","title":"Deep Reinforcement Learning for Multi-Domain Dialogue Systems","date":"2016-11-26","arxiv_id":"1611.08675","repositories_listed":1,"syntology":null},{"url":"/paper/training-an-interactive-humanoid-robot-using","slug":"training-an-interactive-humanoid-robot-using","title":"Training an Interactive Humanoid Robot Using Multimodal Deep Reinforcement Learning","date":"2016-11-26","arxiv_id":"1611.08666","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-object-detection-with-deep","slug":"hierarchical-object-detection-with-deep","title":"Hierarchical Object Detection with Deep Reinforcement Learning","date":"2016-11-11","arxiv_id":"1611.03718","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-in-complex-environments","slug":"learning-to-navigate-in-complex-environments","title":"Learning to Navigate in Complex Environments","date":"2016-11-11","arxiv_id":"1611.03673","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-in-a-day-faster-deep","slug":"learning-to-play-in-a-day-faster-deep","title":"Learning to Play in a Day: Faster Deep Reinforcement Learning by Optimality Tightening","date":"2016-11-05","arxiv_id":"1611.01606","repositories_listed":1,"syntology":null},{"url":"/paper/active-exploration-in-parameterized","slug":"active-exploration-in-parameterized","title":"Active exploration in parameterized reinforcement learning","date":"2016-10-06","arxiv_id":"1610.01986","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-mention","slug":"deep-reinforcement-learning-for-mention","title":"Deep Reinforcement Learning for Mention-Ranking Coreference Models","date":"2016-09-27","arxiv_id":"1609.08667","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-mention#ran","syntology_url":"https://syntology.ai/paper/1609.08667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.08667"}},"official":{"repos":["clarkkev/deep-coref"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/opponent-modeling-in-deep-reinforcement","slug":"opponent-modeling-in-deep-reinforcement","title":"Opponent Modeling in Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05559","repositories_listed":1,"syntology":null},{"url":"/paper/a-threshold-based-scheme-for-reinforcement","slug":"a-threshold-based-scheme-for-reinforcement","title":"A Threshold-based Scheme for Reinforcement Learning in Neural Networks","date":"2016-09-12","arxiv_id":"1609.03348","repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-reinforcement-learning-of","slug":"towards-end-to-end-reinforcement-learning-of","title":"Towards End-to-End Reinforcement Learning of Dialogue Agents for Information Access","date":"2016-09-03","arxiv_id":"1609.00777","repositories_listed":1,"syntology":null},{"url":"/paper/posterior-sampling-for-reinforcement-learning-1","slug":"posterior-sampling-for-reinforcement-learning-1","title":"Posterior Sampling for Reinforcement Learning Without Episodes","date":"2016-08-09","arxiv_id":"1608.02731","repositories_listed":1,"syntology":null},{"url":"/paper/playing-atari-games-with-deep-reinforcement","slug":"playing-atari-games-with-deep-reinforcement","title":"Playing Atari Games with Deep Reinforcement Learning and Human Checkpoint Replay","date":"2016-07-18","arxiv_id":"1607.05077","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-a","slug":"deep-reinforcement-learning-with-a","title":"Deep Reinforcement Learning with a Combinatorial Action Space for Predicting Popular Reddit Threads","date":"2016-06-12","arxiv_id":"1606.03667","repositories_listed":1,"syntology":null},{"url":"/paper/deep-successor-reinforcement-learning","slug":"deep-successor-reinforcement-learning","title":"Deep Successor Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02396","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-successor-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1606.02396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.02396"}},"official":{"repos":["Ardavans/DSR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-end-to-end-learning-for-dialog-state","slug":"towards-end-to-end-learning-for-dialog-state","title":"Towards End-to-End Learning for Dialog State Tracking and Management using Deep Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02560","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-count-based-exploration-and","slug":"unifying-count-based-exploration-and","title":"Unifying Count-Based Exploration and Intrinsic Motivation","date":"2016-06-06","arxiv_id":"1606.01868","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-radio-control-and","slug":"deep-reinforcement-learning-radio-control-and","title":"Deep Reinforcement Learning Radio Control and Signal Detection with KeRLym, a Gym RL Agent","date":"2016-05-30","arxiv_id":"1605.09221","repositories_listed":1,"syntology":null},{"url":"/paper/improving-information-extraction-by-acquiring","slug":"improving-information-extraction-by-acquiring","title":"Improving Information Extraction by Acquiring External Evidence with Reinforcement Learning","date":"2016-03-25","arxiv_id":"1603.07954","repositories_listed":1,"syntology":null},{"url":"/paper/exploratory-gradient-boosting-for","slug":"exploratory-gradient-boosting-for","title":"Exploratory Gradient Boosting for Reinforcement Learning in Complex Domains","date":"2016-03-14","arxiv_id":"1603.04119","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-practical-linear-temporal","slug":"investigating-practical-linear-temporal","title":"Investigating practical linear temporal difference learning","date":"2016-02-28","arxiv_id":"1602.08771","repositories_listed":1,"syntology":null},{"url":"/paper/simpleds-a-simple-deep-reinforcement-learning","slug":"simpleds-a-simple-deep-reinforcement-learning","title":"SimpleDS: A Simple Deep Reinforcement Learning Dialogue System","date":"2016-01-18","arxiv_id":"1601.04574","repositories_listed":1,"syntology":null},{"url":"/paper/angrier-birds-bayesian-reinforcement-learning","slug":"angrier-birds-bayesian-reinforcement-learning","title":"Angrier Birds: Bayesian reinforcement learning","date":"2016-01-06","arxiv_id":"1601.01297","repositories_listed":1,"syntology":null},{"url":"/paper/state-of-the-art-control-of-atari-games-using","slug":"state-of-the-art-control-of-atari-games-using","title":"State of the Art Control of Atari Games Using Shallow Reinforcement Learning","date":"2015-12-04","arxiv_id":"1512.01563","repositories_listed":1,"syntology":null}],"record_sha256":"e6a665d21bd8736c8abf33370fff16c0c90e766a6cd997d780f409ad30f9a059","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}