{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/38","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":38,"pages_in_order":132,"rows_per_page":100,"rows":[3701,3800],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/37","next":"/task/reinforcement-learning/papers/39","papers":[{"url":"/paper/relative-entropy-regularized-policy-iteration","slug":"relative-entropy-regularized-policy-iteration","title":"Relative Entropy Regularized Policy Iteration","date":"2018-12-05","arxiv_id":"1812.02256","repositories_listed":1,"syntology":null},{"url":"/paper/playing-text-adventure-games-with-graph-based","slug":"playing-text-adventure-games-with-graph-based","title":"Playing Text-Adventure Games with Graph-Based Deep Reinforcement Learning","date":"2018-12-04","arxiv_id":"1812.01628","repositories_listed":1,"syntology":null},{"url":"/paper/towards-accurate-task-accomplishment-with-low","slug":"towards-accurate-task-accomplishment-with-low","title":"CRAVES: Controlling Robotic Arm with a Vision-based Economic System","date":"2018-12-03","arxiv_id":"1812.00725","repositories_listed":1,"syntology":null},{"url":"/paper/towards-solving-text-based-games-by-producing","slug":"towards-solving-text-based-games-by-producing","title":"Towards Solving Text-based Games by Producing Adaptive Action Spaces","date":"2018-12-03","arxiv_id":"1812.00855","repositories_listed":1,"syntology":null},{"url":"/paper/visual-foresight-model-based-deep","slug":"visual-foresight-model-based-deep","title":"Visual Foresight: Model-Based Deep Reinforcement Learning for Vision-Based Robotic Control","date":"2018-12-03","arxiv_id":"1812.00568","repositories_listed":1,"syntology":null},{"url":"/paper/macro-action-selection-with-deep","slug":"macro-action-selection-with-deep","title":"Macro action selection with deep reinforcement learning in StarCraft","date":"2018-12-02","arxiv_id":"1812.00336","repositories_listed":1,"syntology":null},{"url":"/paper/data-center-cooling-using-model-predictive","slug":"data-center-cooling-using-model-predictive","title":"Data center cooling using model-predictive control","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exponentially-weighted-imitation-learning-for","slug":"exponentially-weighted-imitation-learning-for","title":"Exponentially Weighted Imitation Learning for Batched Historical Data","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-curriculum-policies-for","slug":"learning-curriculum-policies-for","title":"Learning Curriculum Policies for Reinforcement Learning","date":"2018-12-01","arxiv_id":"1812.00285","repositories_listed":1,"syntology":null},{"url":"/paper/learning-loop-invariants-for-program","slug":"learning-loop-invariants-for-program","title":"Learning Loop Invariants for Program Verification","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/simple-random-search-of-static-linear","slug":"simple-random-search-of-static-linear","title":"Simple random search of static linear policies is competitive for reinforcement learning","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/temporal-regularization-for-markov-decision","slug":"temporal-regularization-for-markov-decision","title":"Temporal Regularization for Markov Decision Process","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/deep-multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1811.12557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.12557"}},"official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-natural-language-emergence-with","slug":"modeling-natural-language-emergence-with","title":"Modeling natural language emergence with integral transform theory and reinforcement learning","date":"2018-11-30","arxiv_id":"1812.01431","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-autonomous","slug":"deep-reinforcement-learning-for-autonomous","title":"Deep Reinforcement Learning for Autonomous Driving","date":"2018-11-28","arxiv_id":"1811.11329","repositories_listed":1,"syntology":null},{"url":"/paper/grammars-and-reinforcement-learning-for","slug":"grammars-and-reinforcement-learning-for","title":"Grammars and reinforcement learning for molecule optimization","date":"2018-11-27","arxiv_id":"1811.11222","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-impact-of-entropy-on-policy","slug":"understanding-the-impact-of-entropy-on-policy","title":"Understanding the impact of entropy on policy optimization","date":"2018-11-27","arxiv_id":"1811.11214","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/understanding-the-impact-of-entropy-on-policy#ran","syntology_url":"https://syntology.ai/paper/1811.11214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.11214"}},"official":{"repos":["zafarali/emdp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-uplift-modeling","slug":"reinforcement-learning-for-uplift-modeling","title":"Reinforcement Learning for Uplift Modeling","date":"2018-11-26","arxiv_id":"1811.10158","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-uplift-modeling#ran","syntology_url":"https://syntology.ai/paper/1811.10158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.10158"}},"official":null}},{"url":"/paper/hardware-conditioned-policies-for-multi-robot","slug":"hardware-conditioned-policies-for-multi-robot","title":"Hardware Conditioned Policies for Multi-Robot Transfer Learning","date":"2018-11-24","arxiv_id":"1811.09864","repositories_listed":1,"syntology":null},{"url":"/paper/hydra-a-peer-to-peer-distributed-training","slug":"hydra-a-peer-to-peer-distributed-training","title":"Hydra: A Peer to Peer Distributed Training & Data Collection Framework","date":"2018-11-24","arxiv_id":"1811.09878","repositories_listed":1,"syntology":null},{"url":"/paper/urban-driving-with-multi-objective-deep","slug":"urban-driving-with-multi-objective-deep","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","date":"2018-11-21","arxiv_id":"1811.08586","repositories_listed":1,"syntology":null},{"url":"/paper/model-learning-for-look-ahead-exploration-in","slug":"model-learning-for-look-ahead-exploration-in","title":"Model Learning for Look-ahead Exploration in Continuous Control","date":"2018-11-20","arxiv_id":"1811.08086","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-of-active-vision-for","slug":"reinforcement-learning-of-active-vision-for","title":"Reinforcement Learning of Active Vision for Manipulating Objects under Occlusions","date":"2018-11-20","arxiv_id":"1811.08067","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-policies-with-language-via-meta","slug":"guiding-policies-with-language-via-meta","title":"Guiding Policies with Language via Meta-Learning","date":"2018-11-19","arxiv_id":"1811.07882","repositories_listed":1,"syntology":null},{"url":"/paper/learning-actionable-representations-with-goal","slug":"learning-actionable-representations-with-goal","title":"Learning Actionable Representations with Goal-Conditioned Policies","date":"2018-11-19","arxiv_id":"1811.07819","repositories_listed":1,"syntology":null},{"url":"/paper/switch-based-active-deep-dyna-q-efficient","slug":"switch-based-active-deep-dyna-q-efficient","title":"Switch-based Active Deep Dyna-Q: Efficient Adaptive Planning for Task-Completion Dialogue Policy Learning","date":"2018-11-19","arxiv_id":"1811.07550","repositories_listed":1,"syntology":null},{"url":"/paper/grasp2vec-learning-object-representations","slug":"grasp2vec-learning-object-representations","title":"Grasp2Vec: Learning Object Representations from Self-Supervised Grasping","date":"2018-11-16","arxiv_id":"1811.06964","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grasp2vec-learning-object-representations#ran","syntology_url":"https://syntology.ai/paper/1811.06964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.06964"}},"official":null}},{"url":"/paper/trolleymod-v10-an-open-source-simulation-and","slug":"trolleymod-v10-an-open-source-simulation-and","title":"TrolleyMod v1.0: An Open-Source Simulation and Data-Collection Platform for Ethical Decision Making in Autonomous Vehicles","date":"2018-11-14","arxiv_id":"1811.05594","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-learning-for-fooling-neural-networks","slug":"deep-q-learning-for-fooling-neural-networks","title":"Deep Q learning for fooling neural networks","date":"2018-11-13","arxiv_id":"1811.05521","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-driven-extensible-hierarchical","slug":"diversity-driven-extensible-hierarchical","title":"Diversity-Driven Extensible Hierarchical Reinforcement Learning","date":"2018-11-10","arxiv_id":"1811.04324","repositories_listed":1,"syntology":null},{"url":"/paper/fully-convolutional-network-with-multi-step","slug":"fully-convolutional-network-with-multi-step","title":"Fully Convolutional Network with Multi-Step Reinforcement Learning for Image Processing","date":"2018-11-10","arxiv_id":"1811.04323","repositories_listed":1,"syntology":null},{"url":"/paper/long-short-term-memory-with-dynamic-skip","slug":"long-short-term-memory-with-dynamic-skip","title":"Long Short-Term Memory with Dynamic Skip Connections","date":"2018-11-09","arxiv_id":"1811.03873","repositories_listed":1,"syntology":null},{"url":"/paper/the-rllchatbot-a-solution-to-the-convai","slug":"the-rllchatbot-a-solution-to-the-convai","title":"The RLLChatbot: a solution to the ConvAI challenge","date":"2018-11-07","arxiv_id":"1811.02714","repositories_listed":1,"syntology":null},{"url":"/paper/ace-an-actor-ensemble-algorithm-for","slug":"ace-an-actor-ensemble-algorithm-for","title":"ACE: An Actor Ensemble Algorithm for Continuous Control with Tree Search","date":"2018-11-06","arxiv_id":"1811.02696","repositories_listed":1,"syntology":null},{"url":"/paper/a-biologically-plausible-learning-rule-for","slug":"a-biologically-plausible-learning-rule-for","title":"A Biologically Plausible Learning Rule for Deep Learning in the Brain","date":"2018-11-05","arxiv_id":"1811.01768","repositories_listed":1,"syntology":null},{"url":"/paper/you-only-search-once-single-shot-neural","slug":"you-only-search-once-single-shot-neural","title":"You Only Search Once: Single Shot Neural Architecture Search via Direct Sparse Optimization","date":"2018-11-05","arxiv_id":"1811.01567","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-action-decoder-for-deep-multi-agent","slug":"bayesian-action-decoder-for-deep-multi-agent","title":"Bayesian Action Decoder for Deep Multi-Agent Reinforcement Learning","date":"2018-11-04","arxiv_id":"1811.01458","repositories_listed":1,"syntology":null},{"url":"/paper/saferoute-learning-to-navigate-streets-safely","slug":"saferoute-learning-to-navigate-streets-safely","title":"SafeRoute: Learning to Navigate Streets Safely in an Urban Environment","date":"2018-11-03","arxiv_id":"1811.01147","repositories_listed":1,"syntology":null},{"url":"/paper/virel-a-variational-inference-framework-for","slug":"virel-a-variational-inference-framework-for","title":"VIREL: A Variational Inference Framework for Reinforcement Learning","date":"2018-11-03","arxiv_id":"1811.01132","repositories_listed":1,"syntology":null},{"url":"/paper/3d-traffic-simulation-for-autonomous-vehicles","slug":"3d-traffic-simulation-for-autonomous-vehicles","title":"3D Traffic Simulation for Autonomous Vehicles in Unity and Python","date":"2018-10-30","arxiv_id":"1810.12552","repositories_listed":1,"syntology":null},{"url":"/paper/gated-hierarchical-attention-for-image","slug":"gated-hierarchical-attention-for-image","title":"Gated Hierarchical Attention for Image Captioning","date":"2018-10-30","arxiv_id":"1810.12535","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-and-deep-learning","slug":"reinforcement-learning-and-deep-learning","title":"Reinforcement Learning and Deep Learning based Lateral Control for Autonomous Driving","date":"2018-10-30","arxiv_id":"1810.12778","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-generalization-in-deep","slug":"assessing-generalization-in-deep","title":"Assessing Generalization in Deep Reinforcement Learning","date":"2018-10-29","arxiv_id":"1810.12282","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assessing-generalization-in-deep#ran","syntology_url":"https://syntology.ai/paper/1810.12282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12282"}},"official":{"repos":["sunblaze-ucb/rl-generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-inference-with-tail-adaptive-f","slug":"variational-inference-with-tail-adaptive-f","title":"Variational Inference with Tail-adaptive f-Divergence","date":"2018-10-29","arxiv_id":"1810.11943","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/variational-inference-with-tail-adaptive-f#ran","syntology_url":"https://syntology.ai/paper/1810.11943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11943"}},"official":{"repos":["dilinwang820/adaptive-f-divergence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dqn-tamer-human-in-the-loop-reinforcement","slug":"dqn-tamer-human-in-the-loop-reinforcement","title":"DQN-TAMER: Human-in-the-Loop Reinforcement Learning with Intractable Feedback","date":"2018-10-28","arxiv_id":"1810.11748","repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-steer-through-deep-reinforcement","slug":"learn-to-steer-through-deep-reinforcement","title":"Learn to Steer through Deep Reinforcement Learning","date":"2018-10-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-common-knowledge-reinforcement","slug":"multi-agent-common-knowledge-reinforcement","title":"Multi-Agent Common Knowledge Reinforcement Learning","date":"2018-10-27","arxiv_id":"1810.11702","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-common-knowledge-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1810.11702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11702"}},"official":{"repos":["schroederdewitt/mackrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transfer-of-deep-reactive-policies-for-mdp","slug":"transfer-of-deep-reactive-policies-for-mdp","title":"Transfer of Deep Reactive Policies for MDP Planning","date":"2018-10-26","arxiv_id":"1810.11488","repositories_listed":1,"syntology":null},{"url":"/paper/inverse-reinforcement-learning-for-video","slug":"inverse-reinforcement-learning-for-video","title":"Inverse reinforcement learning for video games","date":"2018-10-24","arxiv_id":"1810.10593","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-eligibility-traces-for-deep","slug":"efficient-eligibility-traces-for-deep","title":"Reconciling $λ$-Returns with Experience Replay","date":"2018-10-23","arxiv_id":"1810.09967","repositories_listed":1,"syntology":null},{"url":"/paper/actor-critic-policy-optimization-in-partially","slug":"actor-critic-policy-optimization-in-partially","title":"Actor-Critic Policy Optimization in Partially Observable Multiagent Environments","date":"2018-10-21","arxiv_id":"1810.09026","repositories_listed":1,"syntology":null},{"url":"/paper/rlgraph-modular-computation-graphs-for-deep","slug":"rlgraph-modular-computation-graphs-for-deep","title":"RLgraph: Modular Computation Graphs for Deep Reinforcement Learning","date":"2018-10-21","arxiv_id":"1810.09028","repositories_listed":1,"syntology":null},{"url":"/paper/composable-action-conditioned-predictors","slug":"composable-action-conditioned-predictors","title":"Composable Action-Conditioned Predictors: Flexible Off-Policy Learning for Robot Navigation","date":"2018-10-16","arxiv_id":"1810.07167","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-kinematics-and-environment","slug":"integrating-kinematics-and-environment","title":"Integrating kinematics and environment context into deep inverse reinforcement learning for predicting off-road vehicle trajectories","date":"2018-10-16","arxiv_id":"1810.07225","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-decoders-for-fault","slug":"reinforcement-learning-decoders-for-fault","title":"Reinforcement Learning Decoders for Fault-Tolerant Quantum Computation","date":"2018-10-16","arxiv_id":"1810.07207","repositories_listed":1,"syntology":null},{"url":"/paper/social-behavior-learning-with-realistic","slug":"social-behavior-learning-with-realistic","title":"Learning Socially Appropriate Robot Approaching Behavior Toward Groups using Deep Reinforcement Learning","date":"2018-10-16","arxiv_id":"1810.06979","repositories_listed":1,"syntology":null},{"url":"/paper/curious-intrinsically-motivated-multi-task","slug":"curious-intrinsically-motivated-multi-task","title":"CURIOUS: Intrinsically Motivated Modular Multi-Goal Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06284","repositories_listed":1,"syntology":null},{"url":"/paper/deep-transfer-reinforcement-learning-for-text","slug":"deep-transfer-reinforcement-learning-for-text","title":"Deep Transfer Reinforcement Learning for Text Summarization","date":"2018-10-15","arxiv_id":"1810.06667","repositories_listed":1,"syntology":null},{"url":"/paper/factorized-machine-self-confidence-for","slug":"factorized-machine-self-confidence-for","title":"Factorized Machine Self-Confidence for Decision-Making Agents","date":"2018-10-15","arxiv_id":"1810.06519","repositories_listed":1,"syntology":null},{"url":"/paper/multi-stage-reinforcement-learning-for-object","slug":"multi-stage-reinforcement-learning-for-object","title":"Multi-Stage Reinforcement Learning For Object Detection","date":"2018-10-15","arxiv_id":"1810.10325","repositories_listed":1,"syntology":null},{"url":"/paper/visual-semantic-navigation-using-scene-priors","slug":"visual-semantic-navigation-using-scene-priors","title":"Visual Semantic Navigation using Scene Priors","date":"2018-10-15","arxiv_id":"1810.06543","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-potential-of-classical-q","slug":"assessing-the-potential-of-classical-q","title":"Assessing the Potential of Classical Q-learning in General Game Playing","date":"2018-10-14","arxiv_id":"1810.06078","repositories_listed":1,"syntology":null},{"url":"/paper/empowerment-driven-exploration-using-mutual","slug":"empowerment-driven-exploration-using-mutual","title":"Empowerment-driven Exploration using Mutual Information Estimation","date":"2018-10-11","arxiv_id":"1810.05533","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-general-purpose-active-learning","slug":"discovering-general-purpose-active-learning","title":"Discovering General-Purpose Active Learning Strategies","date":"2018-10-09","arxiv_id":"1810.04114","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-improving-agent","slug":"reinforcement-learning-for-improving-agent","title":"Reinforcement Learning for Improving Agent Design","date":"2018-10-09","arxiv_id":"1810.03779","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-deep-reinforcement-learning","slug":"semi-supervised-deep-reinforcement-learning","title":"Semi-supervised Deep Reinforcement Learning in Support of IoT and Smart City Services","date":"2018-10-09","arxiv_id":"1810.04118","repositories_listed":1,"syntology":null},{"url":"/paper/fast-context-adaptation-via-meta-learning","slug":"fast-context-adaptation-via-meta-learning","title":"Fast Context Adaptation via Meta-Learning","date":"2018-10-08","arxiv_id":"1810.03642","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-context-adaptation-via-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1810.03642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.03642"}},"official":{"repos":["lmzintgraf/cavia"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sfv-reinforcement-learning-of-physical-skills","slug":"sfv-reinforcement-learning-of-physical-skills","title":"SFV: Reinforcement Learning of Physical Skills from Videos","date":"2018-10-08","arxiv_id":"1810.03599","repositories_listed":1,"syntology":null},{"url":"/paper/ppo-cma-proximal-policy-optimization-with","slug":"ppo-cma-proximal-policy-optimization-with","title":"PPO-CMA: Proximal Policy Optimization with Covariance Matrix Adaptation","date":"2018-10-05","arxiv_id":"1810.02541","repositories_listed":1,"syntology":null},{"url":"/paper/where-did-my-optimum-go-an-empirical-analysis","slug":"where-did-my-optimum-go-an-empirical-analysis","title":"Where Did My Optimum Go?: An Empirical Analysis of Gradient Descent Optimization in Policy Gradient Methods","date":"2018-10-05","arxiv_id":"1810.02525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/where-did-my-optimum-go-an-empirical-analysis#ran","syntology_url":"https://syntology.ai/paper/1810.02525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02525"}},"official":{"repos":["facebookresearch/WhereDidMyOptimumGo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/episodic-curiosity-through-reachability","slug":"episodic-curiosity-through-reachability","title":"Episodic Curiosity through Reachability","date":"2018-10-04","arxiv_id":"1810.02274","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/episodic-curiosity-through-reachability#ran","syntology_url":"https://syntology.ai/paper/1810.02274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02274"}},"official":{"repos":["google-research/episodic-curiosity"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/image-based-guidance-of-autonomous-aircraft","slug":"image-based-guidance-of-autonomous-aircraft","title":"Image-based Guidance of Autonomous Aircraft for Wildfire Surveillance and Prediction","date":"2018-10-04","arxiv_id":"1810.02455","repositories_listed":1,"syntology":null},{"url":"/paper/mycaffe-a-complete-c-re-write-of-caffe-with","slug":"mycaffe-a-complete-c-re-write-of-caffe-with","title":"MyCaffe: A Complete C# Re-Write of Caffe with Reinforcement Learning","date":"2018-10-04","arxiv_id":"1810.02272","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-skill-composition-and-simulation-to","slug":"zero-shot-skill-composition-and-simulation-to","title":"Simulator Predictive Control: Using Learned Task Representations and MPC for Zero-Shot Generalization and Sequencing","date":"2018-10-04","arxiv_id":"1810.02422","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-of-reinforcement-learning","slug":"comparison-of-reinforcement-learning","title":"Comparison of Reinforcement Learning algorithms applied to the Cart Pole problem","date":"2018-10-03","arxiv_id":"1810.01940","repositories_listed":1,"syntology":null},{"url":"/paper/emi-exploration-with-mutual-information","slug":"emi-exploration-with-mutual-information","title":"EMI: Exploration with Mutual Information","date":"2018-10-02","arxiv_id":"1810.01176","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emi-exploration-with-mutual-information#ran","syntology_url":"https://syntology.ai/paper/1810.01176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01176"}},"official":{"repos":["snu-mllab/EMI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-perturbed-rewards","slug":"reinforcement-learning-with-perturbed-rewards","title":"Reinforcement Learning with Perturbed Rewards","date":"2018-10-02","arxiv_id":"1810.01032","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-perturbed-rewards#ran","syntology_url":"https://syntology.ai/paper/1810.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01032"}},"official":{"repos":["wangjksjtu/rl-perturbed-reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-dreaming-variational-autoencoder-for","slug":"the-dreaming-variational-autoencoder-for","title":"The Dreaming Variational Autoencoder for Reinforcement Learning Environments","date":"2018-10-02","arxiv_id":"1810.01112","repositories_listed":1,"syntology":null},{"url":"/paper/airdialogue-an-environment-for-goal-oriented","slug":"airdialogue-an-environment-for-goal-oriented","title":"AirDialogue: An Environment for Goal-Oriented Dialogue Research","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/automatic-local-rewriting-for-combinatorial","slug":"automatic-local-rewriting-for-combinatorial","title":"Learning to Perform Local Rewriting for Combinatorial Optimization","date":"2018-09-30","arxiv_id":"1810.00337","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-learning-with-corrective-feedback","slug":"interactive-learning-with-corrective-feedback","title":"Interactive Learning with Corrective Feedback for Policies based on Deep Neural Networks","date":"2018-09-30","arxiv_id":"1810.00466","repositories_listed":1,"syntology":null},{"url":"/paper/using-state-predictions-for-value","slug":"using-state-predictions-for-value","title":"Using State Predictions for Value Regularization in Curiosity Driven Deep Reinforcement Learning","date":"2018-09-30","arxiv_id":"1810.00361","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-and-regularization-in-dqn","slug":"generalization-and-regularization-in-dqn","title":"Generalization and Regularization in DQN","date":"2018-09-29","arxiv_id":"1810.00123","repositories_listed":1,"syntology":null},{"url":"/paper/m3rl-mind-aware-multi-agent-management-1","slug":"m3rl-mind-aware-multi-agent-management-1","title":"M$^3$RL: Mind-aware Multi-agent Management Reinforcement Learning","date":"2018-09-29","arxiv_id":"1810.00147","repositories_listed":1,"syntology":null},{"url":"/paper/propagation-networks-for-model-based-control","slug":"propagation-networks-for-model-based-control","title":"Propagation Networks for Model-Based Control Under Partial Observation","date":"2018-09-28","arxiv_id":"1809.11169","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-neural-story-plot-generation-via","slug":"controllable-neural-story-plot-generation-via","title":"Controllable Neural Story Plot Generation via Reward Shaping","date":"2018-09-27","arxiv_id":"1809.10736","repositories_listed":1,"syntology":null},{"url":"/paper/banditsum-extractive-summarization-as-a","slug":"banditsum-extractive-summarization-as-a","title":"BanditSum: Extractive Summarization as a Contextual Bandit","date":"2018-09-25","arxiv_id":"1809.09672","repositories_listed":1,"syntology":null},{"url":"/paper/better-safe-than-sorry-evidence-accumulation","slug":"better-safe-than-sorry-evidence-accumulation","title":"Better Safe than Sorry: Evidence Accumulation Allows for Safe Reinforcement Learning","date":"2018-09-24","arxiv_id":"1809.09147","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-exploration-and-recovery-from","slug":"constrained-exploration-and-recovery-from","title":"Constrained Exploration and Recovery from Experience Shaping","date":"2018-09-21","arxiv_id":"1809.08925","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-contact-forces-for-learning-to","slug":"leveraging-contact-forces-for-learning-to","title":"Leveraging Contact Forces for Learning to Grasp","date":"2018-09-19","arxiv_id":"1809.07004","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-adaptive-optimal-control-of","slug":"model-free-adaptive-optimal-control-of","title":"Model-Free Adaptive Optimal Control of Episodic Fixed-Horizon Manufacturing Processes using Reinforcement Learning","date":"2018-09-18","arxiv_id":"1809.06646","repositories_listed":1,"syntology":null},{"url":"/paper/generalizing-across-multi-objective-reward","slug":"generalizing-across-multi-objective-reward","title":"Generalizing Across Multi-Objective Reward Functions in Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06364","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generalizing-across-multi-objective-reward#ran","syntology_url":"https://syntology.ai/paper/1809.06364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06364"}},"official":null}},{"url":"/paper/muscle-excitation-estimation-in-biomechanical","slug":"muscle-excitation-estimation-in-biomechanical","title":"Muscle Excitation Estimation in Biomechanical Simulation Using NAF Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06121","repositories_listed":1,"syntology":null},{"url":"/paper/transparency-and-explanation-in-deep","slug":"transparency-and-explanation-in-deep","title":"Transparency and Explanation in Deep Reinforcement Learning Neural Networks","date":"2018-09-17","arxiv_id":"1809.06061","repositories_listed":1,"syntology":null},{"url":"/paper/deterministic-implementations-for","slug":"deterministic-implementations-for","title":"Deterministic Implementations for Reproducibility in Deep Reinforcement Learning","date":"2018-09-15","arxiv_id":"1809.05676","repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-interpretability-in-deep-q","slug":"towards-better-interpretability-in-deep-q","title":"Towards Better Interpretability in Deep Q-Networks","date":"2018-09-15","arxiv_id":"1809.05630","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-better-interpretability-in-deep-q#ran","syntology_url":"https://syntology.ai/paper/1809.05630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05630"}},"official":null}},{"url":"/paper/model-based-reinforcement-learning-via-meta","slug":"model-based-reinforcement-learning-via-meta","title":"Model-Based Reinforcement Learning via Meta-Policy Optimization","date":"2018-09-14","arxiv_id":"1809.05214","repositories_listed":1,"syntology":null},{"url":"/paper/online-cyber-attack-detection-in-smart-grid-a","slug":"online-cyber-attack-detection-in-smart-grid-a","title":"Online Cyber-Attack Detection in Smart Grid: A Reinforcement Learning Approach","date":"2018-09-14","arxiv_id":"1809.05258","repositories_listed":1,"syntology":null},{"url":"/paper/cm3-cooperative-multi-goal-multi-stage-multi","slug":"cm3-cooperative-multi-goal-multi-stage-multi","title":"CM3: Cooperative Multi-goal Multi-stage Multi-agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05188","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cm3-cooperative-multi-goal-multi-stage-multi#ran","syntology_url":"https://syntology.ai/paper/1809.05188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05188"}},"official":{"repos":["011235813/cm3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-event","slug":"deep-reinforcement-learning-for-event","title":"Deep Reinforcement Learning for Event-Triggered Control","date":"2018-09-13","arxiv_id":"1809.05152","repositories_listed":1,"syntology":null}],"record_sha256":"dedfa01e426e2357d343a36553a6abc9111f095a0488072113aea027c6093386","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}