{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/42","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":42,"pages_in_order":132,"rows_per_page":100,"rows":[4101,4200],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/41","next":"/task/reinforcement-learning/papers/43","papers":[{"url":"/paper/third-person-imitation-learning","slug":"third-person-imitation-learning","title":"Third-Person Imitation Learning","date":"2017-03-06","arxiv_id":"1703.01703","repositories_listed":1,"syntology":null},{"url":"/paper/count-based-exploration-with-neural-density","slug":"count-based-exploration-with-neural-density","title":"Count-Based Exploration with Neural Density Models","date":"2017-03-03","arxiv_id":"1703.01310","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/count-based-exploration-with-neural-density#ran","syntology_url":"https://syntology.ai/paper/1703.01310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.01310"}},"official":null}},{"url":"/paper/ex2-exploration-with-exemplar-models-for-deep","slug":"ex2-exploration-with-exemplar-models-for-deep","title":"EX2: Exploration with Exemplar Models for Deep Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01260","repositories_listed":1,"syntology":null},{"url":"/paper/feudal-networks-for-hierarchical","slug":"feudal-networks-for-hierarchical","title":"FeUdal Networks for Hierarchical Reinforcement Learning","date":"2017-03-03","arxiv_id":"1703.01161","repositories_listed":1,"syntology":null},{"url":"/paper/generalised-discount-functions-applied-to-a","slug":"generalised-discount-functions-applied-to-a","title":"Generalised Discount Functions applied to a Monte-Carlo AImu Implementation","date":"2017-03-03","arxiv_id":"1703.01358","repositories_listed":1,"syntology":null},{"url":"/paper/a-laplacian-framework-for-option-discovery-in","slug":"a-laplacian-framework-for-option-discovery-in","title":"A Laplacian Framework for Option Discovery in Reinforcement Learning","date":"2017-03-02","arxiv_id":"1703.00956","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-pivoting-task","slug":"reinforcement-learning-for-pivoting-task","title":"Reinforcement Learning for Pivoting Task","date":"2017-03-01","arxiv_id":"1703.00472","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-value-and-policy","slug":"bridging-the-gap-between-value-and-policy","title":"Bridging the Gap Between Value and Policy Based Reinforcement Learning","date":"2017-02-28","arxiv_id":"1702.08892","repositories_listed":1,"syntology":null},{"url":"/paper/neural-map-structured-memory-for-deep","slug":"neural-map-structured-memory-for-deep","title":"Neural Map: Structured Memory for Deep Reinforcement Learning","date":"2017-02-27","arxiv_id":"1702.08360","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-error-propagation-through","slug":"tackling-error-propagation-through","title":"Tackling Error Propagation through Reinforcement Learning: A Case of Greedy Dependency Parsing","date":"2017-02-22","arxiv_id":"1702.06794","repositories_listed":1,"syntology":null},{"url":"/paper/beating-the-worlds-best-at-super-smash-bros","slug":"beating-the-worlds-best-at-super-smash-bros","title":"Beating the World's Best at Super Smash Bros. with Deep Reinforcement Learning","date":"2017-02-21","arxiv_id":"1702.06230","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-visual-tracking-by-deep-reinforced","slug":"real-time-visual-tracking-by-deep-reinforced","title":"Real-time visual tracking by deep reinforced decision making","date":"2017-02-21","arxiv_id":"1702.06291","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-common-implementation-of","slug":"towards-a-common-implementation-of","title":"Towards a Common Implementation of Reinforcement Learning for Multiple Robotic Tasks","date":"2017-02-21","arxiv_id":"1702.06329","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-multi-task-by-active-sampling","slug":"learning-to-multi-task-by-active-sampling","title":"Learning to Multi-Task by Active Sampling","date":"2017-02-20","arxiv_id":"1702.06053","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-deep-reinforcement-learning","slug":"collaborative-deep-reinforcement-learning","title":"Collaborative Deep Reinforcement Learning","date":"2017-02-19","arxiv_id":"1702.05796","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-attacks-on-neural-network","slug":"adversarial-attacks-on-neural-network","title":"Adversarial Attacks on Neural Network Policies","date":"2017-02-08","arxiv_id":"1702.02284","repositories_listed":1,"syntology":null},{"url":"/paper/pathnet-evolution-channels-gradient-descent","slug":"pathnet-evolution-channels-gradient-descent","title":"PathNet: Evolution Channels Gradient Descent in Super Neural Networks","date":"2017-01-30","arxiv_id":"1701.08734","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pathnet-evolution-channels-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/1701.08734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.08734"}},"official":null}},{"url":"/paper/towards-automatic-learning-of-heuristics-for","slug":"towards-automatic-learning-of-heuristics-for","title":"Towards Automatic Learning of Heuristics for Mechanical Transformations of Procedural Code","date":"2017-01-25","arxiv_id":"1701.07123","repositories_listed":1,"syntology":null},{"url":"/paper/vulnerability-of-deep-reinforcement-learning","slug":"vulnerability-of-deep-reinforcement-learning","title":"Vulnerability of Deep Reinforcement Learning to Policy Induction Attacks","date":"2017-01-16","arxiv_id":"1701.04143","repositories_listed":1,"syntology":null},{"url":"/paper/near-optimal-behavior-via-approximate-state","slug":"near-optimal-behavior-via-approximate-state","title":"Near Optimal Behavior via Approximate State Abstraction","date":"2017-01-15","arxiv_id":"1701.04113","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-bidding-by-reinforcement-learning","slug":"real-time-bidding-by-reinforcement-learning","title":"Real-Time Bidding by Reinforcement Learning in Display Advertising","date":"2017-01-10","arxiv_id":"1701.02490","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-via-recurrent","slug":"reinforcement-learning-via-recurrent","title":"Reinforcement Learning via Recurrent Convolutional Neural Networks","date":"2017-01-09","arxiv_id":"1701.02392","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-deep-network-solutions-for","slug":"a-survey-of-deep-network-solutions-for","title":"A Survey of Deep Network Solutions for Learning Control in Robotics: From Reinforcement to Imitation","date":"2016-12-21","arxiv_id":"1612.07139","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-models-for-model-based","slug":"self-correcting-models-for-model-based","title":"Self-Correcting Models for Model-Based Reinforcement Learning","date":"2016-12-19","arxiv_id":"1612.06018","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-optimization-with-robust-bayesian","slug":"bayesian-optimization-with-robust-bayesian","title":"Bayesian Optimization with Robust Bayesian Neural Networks","date":"2016-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/playing-doom-with-slam-augmented-deep","slug":"playing-doom-with-slam-augmented-deep","title":"Playing Doom with SLAM-Augmented Deep Reinforcement Learning","date":"2016-12-01","arxiv_id":"1612.00380","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-domain","slug":"deep-reinforcement-learning-for-multi-domain","title":"Deep Reinforcement Learning for Multi-Domain Dialogue Systems","date":"2016-11-26","arxiv_id":"1611.08675","repositories_listed":1,"syntology":null},{"url":"/paper/training-an-interactive-humanoid-robot-using","slug":"training-an-interactive-humanoid-robot-using","title":"Training an Interactive Humanoid Robot Using Multimodal Deep Reinforcement Learning","date":"2016-11-26","arxiv_id":"1611.08666","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-fast-diverse-decoding-algorithm-for","slug":"a-simple-fast-diverse-decoding-algorithm-for","title":"A Simple, Fast Diverse Decoding Algorithm for Neural Generation","date":"2016-11-25","arxiv_id":"1611.08562","repositories_listed":1,"syntology":null},{"url":"/paper/variational-intrinsic-control","slug":"variational-intrinsic-control","title":"Variational Intrinsic Control","date":"2016-11-22","arxiv_id":"1611.07507","repositories_listed":1,"syntology":null},{"url":"/paper/cad2rl-real-single-image-flight-without-a","slug":"cad2rl-real-single-image-flight-without-a","title":"CAD2RL: Real Single-Image Flight without a Single Real Image","date":"2016-11-13","arxiv_id":"1611.04201","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-object-detection-with-deep","slug":"hierarchical-object-detection-with-deep","title":"Hierarchical Object Detection with Deep Reinforcement Learning","date":"2016-11-11","arxiv_id":"1611.03718","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-in-complex-environments","slug":"learning-to-navigate-in-complex-environments","title":"Learning to Navigate in Complex Environments","date":"2016-11-11","arxiv_id":"1611.03673","repositories_listed":1,"syntology":null},{"url":"/paper/playing-snes-in-the-retro-learning","slug":"playing-snes-in-the-retro-learning","title":"Playing SNES in the Retro Learning Environment","date":"2016-11-07","arxiv_id":"1611.02205","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-in-a-day-faster-deep","slug":"learning-to-play-in-a-day-faster-deep","title":"Learning to Play in a Day: Faster Deep Reinforcement Learning by Optimality Tightening","date":"2016-11-05","arxiv_id":"1611.01606","repositories_listed":1,"syntology":null},{"url":"/paper/dual-learning-for-machine-translation","slug":"dual-learning-for-machine-translation","title":"Dual Learning for Machine Translation","date":"2016-11-01","arxiv_id":"1611.00179","repositories_listed":1,"syntology":null},{"url":"/paper/reset-free-trial-and-error-learning-for-robot","slug":"reset-free-trial-and-error-learning-for-robot","title":"Reset-free Trial-and-Error Learning for Robot Damage Recovery","date":"2016-10-13","arxiv_id":"1610.04213","repositories_listed":1,"syntology":null},{"url":"/paper/active-exploration-in-parameterized","slug":"active-exploration-in-parameterized","title":"Active exploration in parameterized reinforcement learning","date":"2016-10-06","arxiv_id":"1610.01986","repositories_listed":1,"syntology":null},{"url":"/paper/deep-visual-foresight-for-planning-robot","slug":"deep-visual-foresight-for-planning-robot","title":"Deep Visual Foresight for Planning Robot Motion","date":"2016-10-03","arxiv_id":"1610.00696","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-mention","slug":"deep-reinforcement-learning-for-mention","title":"Deep Reinforcement Learning for Mention-Ranking Coreference Models","date":"2016-09-27","arxiv_id":"1609.08667","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-mention#ran","syntology_url":"https://syntology.ai/paper/1609.08667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.08667"}},"official":{"repos":["clarkkev/deep-coref"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/opponent-modeling-in-deep-reinforcement","slug":"opponent-modeling-in-deep-reinforcement","title":"Opponent Modeling in Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05559","repositories_listed":1,"syntology":null},{"url":"/paper/a-threshold-based-scheme-for-reinforcement","slug":"a-threshold-based-scheme-for-reinforcement","title":"A Threshold-based Scheme for Reinforcement Learning in Neural Networks","date":"2016-09-12","arxiv_id":"1609.03348","repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-reinforcement-learning-of","slug":"towards-end-to-end-reinforcement-learning-of","title":"Towards End-to-End Reinforcement Learning of Dialogue Agents for Information Access","date":"2016-09-03","arxiv_id":"1609.00777","repositories_listed":1,"syntology":null},{"url":"/paper/single-shot-adaptive-measurement-for-quantum","slug":"single-shot-adaptive-measurement-for-quantum","title":"Single-shot Adaptive Measurement for Quantum-enhanced Metrology","date":"2016-08-22","arxiv_id":"1608.06238","repositories_listed":1,"syntology":null},{"url":"/paper/posterior-sampling-for-reinforcement-learning-1","slug":"posterior-sampling-for-reinforcement-learning-1","title":"Posterior Sampling for Reinforcement Learning Without Episodes","date":"2016-08-09","arxiv_id":"1608.02731","repositories_listed":1,"syntology":null},{"url":"/paper/playing-atari-games-with-deep-reinforcement","slug":"playing-atari-games-with-deep-reinforcement","title":"Playing Atari Games with Deep Reinforcement Learning and Human Checkpoint Replay","date":"2016-07-18","arxiv_id":"1607.05077","repositories_listed":1,"syntology":null},{"url":"/paper/guided-policy-search-as-approximate-mirror","slug":"guided-policy-search-as-approximate-mirror","title":"Guided Policy Search as Approximate Mirror Descent","date":"2016-07-15","arxiv_id":"1607.04614","repositories_listed":1,"syntology":null},{"url":"/paper/learning-in-quantum-control-high-dimensional","slug":"learning-in-quantum-control-high-dimensional","title":"Learning in Quantum Control: High-Dimensional Global Optimization for Noisy Quantum Dynamics","date":"2016-07-12","arxiv_id":"1607.03428","repositories_listed":1,"syntology":null},{"url":"/paper/actor-critic-versus-direct-policy-search-a","slug":"actor-critic-versus-direct-policy-search-a","title":"Actor-critic versus direct policy search: a comparison based on sample complexity","date":"2016-06-29","arxiv_id":"1606.09152","repositories_listed":1,"syntology":null},{"url":"/paper/safe-exploration-in-finite-markov-decision","slug":"safe-exploration-in-finite-markov-decision","title":"Safe Exploration in Finite Markov Decision Processes with Gaussian Processes","date":"2016-06-15","arxiv_id":"1606.04753","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/safe-exploration-in-finite-markov-decision#ran","syntology_url":"https://syntology.ai/paper/1606.04753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.04753"}},"official":{"repos":["befelix/SafeMDP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-with-a","slug":"deep-reinforcement-learning-with-a","title":"Deep Reinforcement Learning with a Combinatorial Action Space for Predicting Popular Reddit Threads","date":"2016-06-12","arxiv_id":"1606.03667","repositories_listed":1,"syntology":null},{"url":"/paper/deep-successor-reinforcement-learning","slug":"deep-successor-reinforcement-learning","title":"Deep Successor Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02396","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-successor-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1606.02396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.02396"}},"official":{"repos":["Ardavans/DSR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-end-to-end-learning-for-dialog-state","slug":"towards-end-to-end-learning-for-dialog-state","title":"Towards End-to-End Learning for Dialog State Tracking and Management using Deep Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02560","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-count-based-exploration-and","slug":"unifying-count-based-exploration-and","title":"Unifying Count-Based Exploration and Intrinsic Motivation","date":"2016-06-06","arxiv_id":"1606.01868","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-radio-control-and","slug":"deep-reinforcement-learning-radio-control-and","title":"Deep Reinforcement Learning Radio Control and Signal Detection with KeRLym, a Gym RL Agent","date":"2016-05-30","arxiv_id":"1605.09221","repositories_listed":1,"syntology":null},{"url":"/paper/convolutional-neural-networks-for-automatic-1","slug":"convolutional-neural-networks-for-automatic-1","title":"Convolutional Neural Networks For Automatic State-Time Feature Extraction in Reinforcement Learning Applied to Residential Load Control","date":"2016-04-28","arxiv_id":"1604.08382","repositories_listed":1,"syntology":null},{"url":"/paper/improving-information-extraction-by-acquiring","slug":"improving-information-extraction-by-acquiring","title":"Improving Information Extraction by Acquiring External Evidence with Reinforcement Learning","date":"2016-03-25","arxiv_id":"1603.07954","repositories_listed":1,"syntology":null},{"url":"/paper/exploratory-gradient-boosting-for","slug":"exploratory-gradient-boosting-for","title":"Exploratory Gradient Boosting for Reinforcement Learning in Complex Domains","date":"2016-03-14","arxiv_id":"1603.04119","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-practical-linear-temporal","slug":"investigating-practical-linear-temporal","title":"Investigating practical linear temporal difference learning","date":"2016-02-28","arxiv_id":"1602.08771","repositories_listed":1,"syntology":null},{"url":"/paper/simpleds-a-simple-deep-reinforcement-learning","slug":"simpleds-a-simple-deep-reinforcement-learning","title":"SimpleDS: A Simple Deep Reinforcement Learning Dialogue System","date":"2016-01-18","arxiv_id":"1601.04574","repositories_listed":1,"syntology":null},{"url":"/paper/angrier-birds-bayesian-reinforcement-learning","slug":"angrier-birds-bayesian-reinforcement-learning","title":"Angrier Birds: Bayesian reinforcement learning","date":"2016-01-06","arxiv_id":"1601.01297","repositories_listed":1,"syntology":null},{"url":"/paper/true-online-temporal-difference-learning","slug":"true-online-temporal-difference-learning","title":"True Online Temporal-Difference Learning","date":"2015-12-13","arxiv_id":"1512.04087","repositories_listed":1,"syntology":null},{"url":"/paper/state-of-the-art-control-of-atari-games-using","slug":"state-of-the-art-control-of-atari-games-using","title":"State of the Art Control of Atari Games Using Shallow Reinforcement Learning","date":"2015-12-04","arxiv_id":"1512.01563","repositories_listed":1,"syntology":null},{"url":"/paper/strategic-dialogue-management-via-deep","slug":"strategic-dialogue-management-via-deep","title":"Strategic Dialogue Management via Deep Reinforcement Learning","date":"2015-11-25","arxiv_id":"1511.08099","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-computation-in-neural-networks","slug":"conditional-computation-in-neural-networks","title":"Conditional Computation in Neural Networks for faster models","date":"2015-11-19","arxiv_id":"1511.06297","repositories_listed":1,"syntology":null},{"url":"/paper/policy-distillation","slug":"policy-distillation","title":"Policy Distillation","date":"2015-11-19","arxiv_id":"1511.06295","repositories_listed":1,"syntology":null},{"url":"/paper/deep-spatial-autoencoders-for-visuomotor","slug":"deep-spatial-autoencoders-for-visuomotor","title":"Deep Spatial Autoencoders for Visuomotor Learning","date":"2015-09-21","arxiv_id":"1509.06113","repositories_listed":1,"syntology":null},{"url":"/paper/action-conditional-video-prediction-using","slug":"action-conditional-video-prediction-using","title":"Action-Conditional Video Prediction using Deep Networks in Atari Games","date":"2015-07-31","arxiv_id":"1507.08750","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-exploration-in-reinforcement","slug":"incentivizing-exploration-in-reinforcement","title":"Incentivizing Exploration In Reinforcement Learning With Deep Predictive Models","date":"2015-07-03","arxiv_id":"1507.00814","repositories_listed":1,"syntology":null},{"url":"/paper/gradient-estimation-using-stochastic","slug":"gradient-estimation-using-stochastic","title":"Gradient Estimation Using Stochastic Computation Graphs","date":"2015-06-17","arxiv_id":"1506.05254","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-neural-turing-machines","slug":"reinforcement-learning-neural-turing-machines","title":"Reinforcement Learning Neural Turing Machines - Revised","date":"2015-05-04","arxiv_id":"1505.00521","repositories_listed":1,"syntology":null},{"url":"/paper/gaussian-processes-for-data-efficient","slug":"gaussian-processes-for-data-efficient","title":"Gaussian Processes for Data-Efficient Learning in Robotics and Control","date":"2015-02-10","arxiv_id":"1502.02860","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-in-neural-networks-an-overview","slug":"deep-learning-in-neural-networks-an-overview","title":"Deep Learning in Neural Networks: An Overview","date":"2014-04-30","arxiv_id":"1404.7828","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-the-cvar-via-sampling","slug":"optimizing-the-cvar-via-sampling","title":"Optimizing the CVaR via Sampling","date":"2014-04-15","arxiv_id":"1404.3862","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-planning-and-learning-for-multiagent","slug":"scalable-planning-and-learning-for-multiagent","title":"Scalable Planning and Learning for Multiagent POMDPs: Extended Version","date":"2014-04-04","arxiv_id":"1404.1140","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-general-value-functions-to","slug":"off-policy-general-value-functions-to","title":"Off-Policy General Value Functions to Represent Dynamic Role Assignments in RoboCup 3D Soccer Simulation","date":"2014-02-18","arxiv_id":"1402.4525","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-and-exploration-via-randomized","slug":"generalization-and-exploration-via-randomized","title":"Generalization and Exploration via Randomized Value Functions","date":"2014-02-04","arxiv_id":"1402.0635","repositories_listed":1,"syntology":null},{"url":"/paper/using-reinforcement-learning-to-find-an","slug":"using-reinforcement-learning-to-find-an","title":"Using reinforcement learning to find an optimal set of features","date":"2013-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-actor-critic","slug":"off-policy-actor-critic","title":"Off-Policy Actor-Critic","date":"2012-05-22","arxiv_id":"1205.4839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-policy-actor-critic#ran","syntology_url":"https://syntology.ai/paper/1205.4839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1205.4839"}},"official":null}},{"url":"/paper/nonlinear-inverse-reinforcement-learning-with","slug":"nonlinear-inverse-reinforcement-learning-with","title":"Nonlinear Inverse Reinforcement Learning with Gaussian Processes","date":"2011-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-object-oriented-representation-for","slug":"an-object-oriented-representation-for","title":"An Object-Oriented Representation for Efficient Reinforcement Learning","date":"2008-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/between-mdps-and-semi-mdps-a-framework-for","slug":"between-mdps-and-semi-mdps-a-framework-for","title":"Between MDPs and semi-MDPs: A framework for temporal abstraction in reinforcement learning","date":"1999-08-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/simple-statistical-gradient-following","slug":"simple-statistical-gradient-following","title":"Simple Statistical Gradient-Following Algorithms for Connectionist Reinforcement Learning","date":"1992-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"cuda-l1-improving-cuda-optimization-via","title":"CUDA-L1: Improving CUDA Optimization via Contrastive Reinforcement Learning","date":"2025-07-18","arxiv_id":"2507.14111","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-humans-and-robots-via-reinforcement","title":"Aligning Humans and Robots via Reinforcement Learning from Implicit Human Feedback","date":"2025-07-17","arxiv_id":"2507.13171","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-resource-management-in-1","title":"Autonomous Resource Management in Microservice Systems via Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.12879","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-novelty-to-imitation-self-distilled","title":"From Novelty to Imitation: Self-Distilled Rewards for Offline Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.12815","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-bellman-method-unifying","title":"Spectral Bellman Method: Unifying Representation and Exploration in RL","date":"2025-07-17","arxiv_id":"2507.13181","repositories_listed":0,"syntology":null},{"url":null,"slug":"visionthink-smart-and-efficient-vision","title":"VisionThink: Smart and Efficient Vision Language Model via Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.13348","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-explainable-reinforcement-1","title":"A Survey of Explainable Reinforcement Learning: Targets, Methods and Needs","date":"2025-07-16","arxiv_id":"2507.12599","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-on-path","title":"Distributional Reinforcement Learning on Path-dependent Options","date":"2025-07-16","arxiv_id":"2507.12657","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-reinforcement-learning-sample","title":"Improving Reinforcement Learning Sample-Efficiency using Local Approximation","date":"2025-07-16","arxiv_id":"2507.12383","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-rl-unlocking-diverse-reasoning-in","title":"Scaling Up RL: Unlocking Diverse Reasoning in LLMs via Prolonged Training","date":"2025-07-16","arxiv_id":"2507.12507","repositories_listed":0,"syntology":null},{"url":null,"slug":"thought-purity-defense-paradigm-for-chain-of","title":"Thought Purity: Defense Paradigm For Chain-of-Thought Attack","date":"2025-07-16","arxiv_id":"2507.12314","repositories_listed":0,"syntology":null},{"url":null,"slug":"functional-emotion-modeling-in-biomimetic","title":"Functional Emotion Modeling in Biomimetic Reinforcement Learning","date":"2025-07-15","arxiv_id":"2507.11027","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-pairwise-distance-matching-for","title":"Local Pairwise Distance Matching for Backpropagation-Free Reinforcement Learning","date":"2025-07-15","arxiv_id":"2507.11367","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactical-decision-for-multi-ugv-confrontation","title":"Tactical Decision for Multi-UGV Confrontation with a Vision-Language Model-Based Commander","date":"2025-07-15","arxiv_id":"2507.11079","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-reinforcement-learning-by-planning","title":"Continual Reinforcement Learning by Planning with Online World Models","date":"2025-07-12","arxiv_id":"2507.09177","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrls-chain-of-thought-reasoning-via-latent","title":"CTRLS: Chain-of-Thought Reasoning via Latent State-Transition","date":"2025-07-10","arxiv_id":"2507.08182","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-curiosity-to-competence-how-world-models","title":"From Curiosity to Competence: How World Models Interact with the Dynamics of Exploration","date":"2025-07-10","arxiv_id":"2507.08210","repositories_listed":0,"syntology":null}],"record_sha256":"7a30dec6295d07b207266139b75547ce84be7c4534bb9e15987799c105e9154a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}