{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/39","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":39,"pages_in_order":135,"rows_per_page":100,"rows":[3801,3900],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/38","next":"/task/reinforcement-learning-2/papers/40","papers":[{"url":"/paper/playing-text-adventure-games-with-graph-based","slug":"playing-text-adventure-games-with-graph-based","title":"Playing Text-Adventure Games with Graph-Based Deep Reinforcement Learning","date":"2018-12-04","arxiv_id":"1812.01628","repositories_listed":1,"syntology":null},{"url":"/paper/towards-solving-text-based-games-by-producing","slug":"towards-solving-text-based-games-by-producing","title":"Towards Solving Text-based Games by Producing Adaptive Action Spaces","date":"2018-12-03","arxiv_id":"1812.00855","repositories_listed":1,"syntology":null},{"url":"/paper/visual-foresight-model-based-deep","slug":"visual-foresight-model-based-deep","title":"Visual Foresight: Model-Based Deep Reinforcement Learning for Vision-Based Robotic Control","date":"2018-12-03","arxiv_id":"1812.00568","repositories_listed":1,"syntology":null},{"url":"/paper/macro-action-selection-with-deep","slug":"macro-action-selection-with-deep","title":"Macro action selection with deep reinforcement learning in StarCraft","date":"2018-12-02","arxiv_id":"1812.00336","repositories_listed":1,"syntology":null},{"url":"/paper/data-center-cooling-using-model-predictive","slug":"data-center-cooling-using-model-predictive","title":"Data center cooling using model-predictive control","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exponentially-weighted-imitation-learning-for","slug":"exponentially-weighted-imitation-learning-for","title":"Exponentially Weighted Imitation Learning for Batched Historical Data","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-curriculum-policies-for","slug":"learning-curriculum-policies-for","title":"Learning Curriculum Policies for Reinforcement Learning","date":"2018-12-01","arxiv_id":"1812.00285","repositories_listed":1,"syntology":null},{"url":"/paper/simple-random-search-of-static-linear","slug":"simple-random-search-of-static-linear","title":"Simple random search of static linear policies is competitive for reinforcement learning","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/temporal-regularization-for-markov-decision","slug":"temporal-regularization-for-markov-decision","title":"Temporal Regularization for Markov Decision Process","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/deep-multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1811.12557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.12557"}},"official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-natural-language-emergence-with","slug":"modeling-natural-language-emergence-with","title":"Modeling natural language emergence with integral transform theory and reinforcement learning","date":"2018-11-30","arxiv_id":"1812.01431","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-autonomous","slug":"deep-reinforcement-learning-for-autonomous","title":"Deep Reinforcement Learning for Autonomous Driving","date":"2018-11-28","arxiv_id":"1811.11329","repositories_listed":1,"syntology":null},{"url":"/paper/grammars-and-reinforcement-learning-for","slug":"grammars-and-reinforcement-learning-for","title":"Grammars and reinforcement learning for molecule optimization","date":"2018-11-27","arxiv_id":"1811.11222","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-impact-of-entropy-on-policy","slug":"understanding-the-impact-of-entropy-on-policy","title":"Understanding the impact of entropy on policy optimization","date":"2018-11-27","arxiv_id":"1811.11214","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/understanding-the-impact-of-entropy-on-policy#ran","syntology_url":"https://syntology.ai/paper/1811.11214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.11214"}},"official":{"repos":["zafarali/emdp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-uplift-modeling","slug":"reinforcement-learning-for-uplift-modeling","title":"Reinforcement Learning for Uplift Modeling","date":"2018-11-26","arxiv_id":"1811.10158","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-uplift-modeling#ran","syntology_url":"https://syntology.ai/paper/1811.10158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.10158"}},"official":null}},{"url":"/paper/urban-driving-with-multi-objective-deep","slug":"urban-driving-with-multi-objective-deep","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","date":"2018-11-21","arxiv_id":"1811.08586","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-of-active-vision-for","slug":"reinforcement-learning-of-active-vision-for","title":"Reinforcement Learning of Active Vision for Manipulating Objects under Occlusions","date":"2018-11-20","arxiv_id":"1811.08067","repositories_listed":1,"syntology":null},{"url":"/paper/learning-actionable-representations-with-goal","slug":"learning-actionable-representations-with-goal","title":"Learning Actionable Representations with Goal-Conditioned Policies","date":"2018-11-19","arxiv_id":"1811.07819","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-driven-extensible-hierarchical","slug":"diversity-driven-extensible-hierarchical","title":"Diversity-Driven Extensible Hierarchical Reinforcement Learning","date":"2018-11-10","arxiv_id":"1811.04324","repositories_listed":1,"syntology":null},{"url":"/paper/fully-convolutional-network-with-multi-step","slug":"fully-convolutional-network-with-multi-step","title":"Fully Convolutional Network with Multi-Step Reinforcement Learning for Image Processing","date":"2018-11-10","arxiv_id":"1811.04323","repositories_listed":1,"syntology":null},{"url":"/paper/ace-an-actor-ensemble-algorithm-for","slug":"ace-an-actor-ensemble-algorithm-for","title":"ACE: An Actor Ensemble Algorithm for Continuous Control with Tree Search","date":"2018-11-06","arxiv_id":"1811.02696","repositories_listed":1,"syntology":null},{"url":"/paper/a-biologically-plausible-learning-rule-for","slug":"a-biologically-plausible-learning-rule-for","title":"A Biologically Plausible Learning Rule for Deep Learning in the Brain","date":"2018-11-05","arxiv_id":"1811.01768","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-action-decoder-for-deep-multi-agent","slug":"bayesian-action-decoder-for-deep-multi-agent","title":"Bayesian Action Decoder for Deep Multi-Agent Reinforcement Learning","date":"2018-11-04","arxiv_id":"1811.01458","repositories_listed":1,"syntology":null},{"url":"/paper/virel-a-variational-inference-framework-for","slug":"virel-a-variational-inference-framework-for","title":"VIREL: A Variational Inference Framework for Reinforcement Learning","date":"2018-11-03","arxiv_id":"1811.01132","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-and-deep-learning","slug":"reinforcement-learning-and-deep-learning","title":"Reinforcement Learning and Deep Learning based Lateral Control for Autonomous Driving","date":"2018-10-30","arxiv_id":"1810.12778","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-generalization-in-deep","slug":"assessing-generalization-in-deep","title":"Assessing Generalization in Deep Reinforcement Learning","date":"2018-10-29","arxiv_id":"1810.12282","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assessing-generalization-in-deep#ran","syntology_url":"https://syntology.ai/paper/1810.12282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12282"}},"official":{"repos":["sunblaze-ucb/rl-generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dqn-tamer-human-in-the-loop-reinforcement","slug":"dqn-tamer-human-in-the-loop-reinforcement","title":"DQN-TAMER: Human-in-the-Loop Reinforcement Learning with Intractable Feedback","date":"2018-10-28","arxiv_id":"1810.11748","repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-steer-through-deep-reinforcement","slug":"learn-to-steer-through-deep-reinforcement","title":"Learn to Steer through Deep Reinforcement Learning","date":"2018-10-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-common-knowledge-reinforcement","slug":"multi-agent-common-knowledge-reinforcement","title":"Multi-Agent Common Knowledge Reinforcement Learning","date":"2018-10-27","arxiv_id":"1810.11702","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-common-knowledge-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1810.11702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11702"}},"official":{"repos":["schroederdewitt/mackrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-reinforcement-learning-for-video","slug":"inverse-reinforcement-learning-for-video","title":"Inverse reinforcement learning for video games","date":"2018-10-24","arxiv_id":"1810.10593","repositories_listed":1,"syntology":null},{"url":"/paper/actor-critic-policy-optimization-in-partially","slug":"actor-critic-policy-optimization-in-partially","title":"Actor-Critic Policy Optimization in Partially Observable Multiagent Environments","date":"2018-10-21","arxiv_id":"1810.09026","repositories_listed":1,"syntology":null},{"url":"/paper/rlgraph-modular-computation-graphs-for-deep","slug":"rlgraph-modular-computation-graphs-for-deep","title":"RLgraph: Modular Computation Graphs for Deep Reinforcement Learning","date":"2018-10-21","arxiv_id":"1810.09028","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-decoders-for-fault","slug":"reinforcement-learning-decoders-for-fault","title":"Reinforcement Learning Decoders for Fault-Tolerant Quantum Computation","date":"2018-10-16","arxiv_id":"1810.07207","repositories_listed":1,"syntology":null},{"url":"/paper/social-behavior-learning-with-realistic","slug":"social-behavior-learning-with-realistic","title":"Learning Socially Appropriate Robot Approaching Behavior Toward Groups using Deep Reinforcement Learning","date":"2018-10-16","arxiv_id":"1810.06979","repositories_listed":1,"syntology":null},{"url":"/paper/curious-intrinsically-motivated-multi-task","slug":"curious-intrinsically-motivated-multi-task","title":"CURIOUS: Intrinsically Motivated Modular Multi-Goal Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06284","repositories_listed":1,"syntology":null},{"url":"/paper/deep-transfer-reinforcement-learning-for-text","slug":"deep-transfer-reinforcement-learning-for-text","title":"Deep Transfer Reinforcement Learning for Text Summarization","date":"2018-10-15","arxiv_id":"1810.06667","repositories_listed":1,"syntology":null},{"url":"/paper/multi-stage-reinforcement-learning-for-object","slug":"multi-stage-reinforcement-learning-for-object","title":"Multi-Stage Reinforcement Learning For Object Detection","date":"2018-10-15","arxiv_id":"1810.10325","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-potential-of-classical-q","slug":"assessing-the-potential-of-classical-q","title":"Assessing the Potential of Classical Q-learning in General Game Playing","date":"2018-10-14","arxiv_id":"1810.06078","repositories_listed":1,"syntology":null},{"url":"/paper/empowerment-driven-exploration-using-mutual","slug":"empowerment-driven-exploration-using-mutual","title":"Empowerment-driven Exploration using Mutual Information Estimation","date":"2018-10-11","arxiv_id":"1810.05533","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-general-purpose-active-learning","slug":"discovering-general-purpose-active-learning","title":"Discovering General-Purpose Active Learning Strategies","date":"2018-10-09","arxiv_id":"1810.04114","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-improving-agent","slug":"reinforcement-learning-for-improving-agent","title":"Reinforcement Learning for Improving Agent Design","date":"2018-10-09","arxiv_id":"1810.03779","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-deep-reinforcement-learning","slug":"semi-supervised-deep-reinforcement-learning","title":"Semi-supervised Deep Reinforcement Learning in Support of IoT and Smart City Services","date":"2018-10-09","arxiv_id":"1810.04118","repositories_listed":1,"syntology":null},{"url":"/paper/fast-context-adaptation-via-meta-learning","slug":"fast-context-adaptation-via-meta-learning","title":"Fast Context Adaptation via Meta-Learning","date":"2018-10-08","arxiv_id":"1810.03642","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-context-adaptation-via-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1810.03642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.03642"}},"official":{"repos":["lmzintgraf/cavia"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sfv-reinforcement-learning-of-physical-skills","slug":"sfv-reinforcement-learning-of-physical-skills","title":"SFV: Reinforcement Learning of Physical Skills from Videos","date":"2018-10-08","arxiv_id":"1810.03599","repositories_listed":1,"syntology":null},{"url":"/paper/mycaffe-a-complete-c-re-write-of-caffe-with","slug":"mycaffe-a-complete-c-re-write-of-caffe-with","title":"MyCaffe: A Complete C# Re-Write of Caffe with Reinforcement Learning","date":"2018-10-04","arxiv_id":"1810.02272","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-of-reinforcement-learning","slug":"comparison-of-reinforcement-learning","title":"Comparison of Reinforcement Learning algorithms applied to the Cart Pole problem","date":"2018-10-03","arxiv_id":"1810.01940","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-perturbed-rewards","slug":"reinforcement-learning-with-perturbed-rewards","title":"Reinforcement Learning with Perturbed Rewards","date":"2018-10-02","arxiv_id":"1810.01032","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-perturbed-rewards#ran","syntology_url":"https://syntology.ai/paper/1810.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01032"}},"official":{"repos":["wangjksjtu/rl-perturbed-reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-dreaming-variational-autoencoder-for","slug":"the-dreaming-variational-autoencoder-for","title":"The Dreaming Variational Autoencoder for Reinforcement Learning Environments","date":"2018-10-02","arxiv_id":"1810.01112","repositories_listed":1,"syntology":null},{"url":"/paper/using-state-predictions-for-value","slug":"using-state-predictions-for-value","title":"Using State Predictions for Value Regularization in Curiosity Driven Deep Reinforcement Learning","date":"2018-09-30","arxiv_id":"1810.00361","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-and-regularization-in-dqn","slug":"generalization-and-regularization-in-dqn","title":"Generalization and Regularization in DQN","date":"2018-09-29","arxiv_id":"1810.00123","repositories_listed":1,"syntology":null},{"url":"/paper/m3rl-mind-aware-multi-agent-management-1","slug":"m3rl-mind-aware-multi-agent-management-1","title":"M$^3$RL: Mind-aware Multi-agent Management Reinforcement Learning","date":"2018-09-29","arxiv_id":"1810.00147","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-neural-story-plot-generation-via","slug":"controllable-neural-story-plot-generation-via","title":"Controllable Neural Story Plot Generation via Reward Shaping","date":"2018-09-27","arxiv_id":"1809.10736","repositories_listed":1,"syntology":null},{"url":"/paper/better-safe-than-sorry-evidence-accumulation","slug":"better-safe-than-sorry-evidence-accumulation","title":"Better Safe than Sorry: Evidence Accumulation Allows for Safe Reinforcement Learning","date":"2018-09-24","arxiv_id":"1809.09147","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-exploration-and-recovery-from","slug":"constrained-exploration-and-recovery-from","title":"Constrained Exploration and Recovery from Experience Shaping","date":"2018-09-21","arxiv_id":"1809.08925","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-adaptive-optimal-control-of","slug":"model-free-adaptive-optimal-control-of","title":"Model-Free Adaptive Optimal Control of Episodic Fixed-Horizon Manufacturing Processes using Reinforcement Learning","date":"2018-09-18","arxiv_id":"1809.06646","repositories_listed":1,"syntology":null},{"url":"/paper/generalizing-across-multi-objective-reward","slug":"generalizing-across-multi-objective-reward","title":"Generalizing Across Multi-Objective Reward Functions in Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06364","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generalizing-across-multi-objective-reward#ran","syntology_url":"https://syntology.ai/paper/1809.06364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06364"}},"official":null}},{"url":"/paper/muscle-excitation-estimation-in-biomechanical","slug":"muscle-excitation-estimation-in-biomechanical","title":"Muscle Excitation Estimation in Biomechanical Simulation Using NAF Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06121","repositories_listed":1,"syntology":null},{"url":"/paper/transparency-and-explanation-in-deep","slug":"transparency-and-explanation-in-deep","title":"Transparency and Explanation in Deep Reinforcement Learning Neural Networks","date":"2018-09-17","arxiv_id":"1809.06061","repositories_listed":1,"syntology":null},{"url":"/paper/deterministic-implementations-for","slug":"deterministic-implementations-for","title":"Deterministic Implementations for Reproducibility in Deep Reinforcement Learning","date":"2018-09-15","arxiv_id":"1809.05676","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-via-meta","slug":"model-based-reinforcement-learning-via-meta","title":"Model-Based Reinforcement Learning via Meta-Policy Optimization","date":"2018-09-14","arxiv_id":"1809.05214","repositories_listed":1,"syntology":null},{"url":"/paper/online-cyber-attack-detection-in-smart-grid-a","slug":"online-cyber-attack-detection-in-smart-grid-a","title":"Online Cyber-Attack Detection in Smart Grid: A Reinforcement Learning Approach","date":"2018-09-14","arxiv_id":"1809.05258","repositories_listed":1,"syntology":null},{"url":"/paper/cm3-cooperative-multi-goal-multi-stage-multi","slug":"cm3-cooperative-multi-goal-multi-stage-multi","title":"CM3: Cooperative Multi-goal Multi-stage Multi-agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05188","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cm3-cooperative-multi-goal-multi-stage-multi#ran","syntology_url":"https://syntology.ai/paper/1809.05188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05188"}},"official":{"repos":["011235813/cm3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-event","slug":"deep-reinforcement-learning-for-event","title":"Deep Reinforcement Learning for Event-Triggered Control","date":"2018-09-13","arxiv_id":"1809.05152","repositories_listed":1,"syntology":null},{"url":"/paper/improving-reinforcement-learning-based-image","slug":"improving-reinforcement-learning-based-image","title":"Improving Reinforcement Learning Based Image Captioning with Natural Language Prior","date":"2018-09-13","arxiv_id":"1809.06227","repositories_listed":1,"syntology":null},{"url":"/paper/negative-update-intervals-in-deep-multi-agent","slug":"negative-update-intervals-in-deep-multi-agent","title":"Negative Update Intervals in Deep Multi-Agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05096","repositories_listed":1,"syntology":null},{"url":"/paper/combined-reinforcement-learning-via-abstract","slug":"combined-reinforcement-learning-via-abstract","title":"Combined Reinforcement Learning via Abstract Representations","date":"2018-09-12","arxiv_id":"1809.04506","repositories_listed":1,"syntology":null},{"url":"/paper/sai-a-sensible-artificial-intelligence-that","slug":"sai-a-sensible-artificial-intelligence-that","title":"SAI, a Sensible Artificial Intelligence that plays Go","date":"2018-09-11","arxiv_id":"1809.03928","repositories_listed":1,"syntology":null},{"url":"/paper/improving-optimization-bounds-using-machine","slug":"improving-optimization-bounds-using-machine","title":"Improving Optimization Bounds using Machine Learning: Decision Diagrams meet Deep Reinforcement Learning","date":"2018-09-10","arxiv_id":"1809.03359","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-optimization-bounds-using-machine#ran","syntology_url":"https://syntology.ai/paper/1809.03359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.03359"}},"official":{"repos":["qcappart/learning-DD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/keep-it-stupid-simple","slug":"keep-it-stupid-simple","title":"Combining imagination and heuristics to learn strategies that generalize","date":"2018-09-10","arxiv_id":"1809.03406","repositories_listed":1,"syntology":null},{"url":"/paper/learning-invariances-for-policy","slug":"learning-invariances-for-policy","title":"Learning Invariances for Policy Generalization","date":"2018-09-07","arxiv_id":"1809.02591","repositories_listed":1,"syntology":null},{"url":"/paper/accelerated-reinforcement-learning-for","slug":"accelerated-reinforcement-learning-for","title":"Accelerated Reinforcement Learning for Sentence Generation by Vocabulary Prediction","date":"2018-09-05","arxiv_id":"1809.01694","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-under-threats","slug":"reinforcement-learning-under-threats","title":"Reinforcement Learning under Threats","date":"2018-09-05","arxiv_id":"1809.01560","repositories_listed":1,"syntology":null},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exit-oos-towards-learning-from-planning-in","slug":"exit-oos-towards-learning-from-planning-in","title":"ExIt-OOS: Towards Learning from Planning in Imperfect Information Games","date":"2018-08-30","arxiv_id":"1808.10120","repositories_listed":1,"syntology":null},{"url":"/paper/april-interactively-learning-to-summarise-by","slug":"april-interactively-learning-to-summarise-by","title":"APRIL: Interactively Learning to Summarise by Combining Active Preference Learning and Reinforcement Learning","date":"2018-08-29","arxiv_id":"1808.09658","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/april-interactively-learning-to-summarise-by#ran","syntology_url":"https://syntology.ai/paper/1808.09658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09658"}},"official":{"repos":["UKPLab/emnlp2018-april"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cycle-of-learning-for-autonomous-systems-from","slug":"cycle-of-learning-for-autonomous-systems-from","title":"Cycle-of-Learning for Autonomous Systems from Human Interaction","date":"2018-08-28","arxiv_id":"1808.09572","repositories_listed":1,"syntology":null},{"url":"/paper/solar-deep-structured-representations-for","slug":"solar-deep-structured-representations-for","title":"SOLAR: Deep Structured Representations for Model-Based Reinforcement Learning","date":"2018-08-28","arxiv_id":"1808.09105","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-reinforcement-learning-for-neural","slug":"a-study-of-reinforcement-learning-for-neural","title":"A Study of Reinforcement Learning for Neural Machine Translation","date":"2018-08-27","arxiv_id":"1808.08866","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-study-of-reinforcement-learning-for-neural#ran","syntology_url":"https://syntology.ai/paper/1808.08866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.08866"}},"official":{"repos":["apeterswu/RL4NMT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-semantic-parsing-for-if-then","slug":"interactive-semantic-parsing-for-if-then","title":"Interactive Semantic Parsing for If-Then Recipes via Hierarchical Reinforcement Learning","date":"2018-08-21","arxiv_id":"1808.06740","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-deep-reinforcement-learning","slug":"an-efficient-deep-reinforcement-learning","title":"An Efficient Deep Reinforcement Learning Model for Urban Traffic Control","date":"2018-08-06","arxiv_id":"1808.01876","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-share-and-hide-intentions-using","slug":"learning-to-share-and-hide-intentions-using","title":"Learning to Share and Hide Intentions using Information Regularization","date":"2018-08-06","arxiv_id":"1808.02093","repositories_listed":1,"syntology":null},{"url":"/paper/recogym-a-reinforcement-learning-environment","slug":"recogym-a-reinforcement-learning-environment","title":"RecoGym: A Reinforcement Learning Environment for the problem of Product Recommendation in Online Advertising","date":"2018-08-02","arxiv_id":"1808.00720","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recogym-a-reinforcement-learning-environment#ran","syntology_url":"https://syntology.ai/paper/1808.00720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.00720"}},"official":{"repos":["criteo-research/reco-gym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distantly-supervised-ner-with-partial","slug":"distantly-supervised-ner-with-partial","title":"Distantly Supervised NER with Partial Annotation Learning and Reinforcement Learning","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-1","slug":"multi-agent-deep-reinforcement-learning-for-1","title":"Multi-Agent Deep Reinforcement Learning for Dynamic Power Allocation in Wireless Networks","date":"2018-08-01","arxiv_id":"1808.00490","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-generative-adversarial-imitation","slug":"multi-agent-generative-adversarial-imitation","title":"Multi-Agent Generative Adversarial Imitation Learning","date":"2018-07-26","arxiv_id":"1807.09936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-generative-adversarial-imitation#ran","syntology_url":"https://syntology.ai/paper/1807.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.09936"}},"official":null}},{"url":"/paper/multi-agent-reinforcement-learning-a-report","slug":"multi-agent-reinforcement-learning-a-report","title":"Multi-Agent Reinforcement Learning: A Report on Challenges and Approaches","date":"2018-07-25","arxiv_id":"1807.09427","repositories_listed":1,"syntology":null},{"url":"/paper/learning-heuristics-for-automated-reasoning","slug":"learning-heuristics-for-automated-reasoning","title":"Learning Heuristics for Quantified Boolean Formulas through Deep Reinforcement Learning","date":"2018-07-20","arxiv_id":"1807.08058","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-for-zero","slug":"hierarchical-reinforcement-learning-for-zero","title":"Hierarchical Reinforcement Learning for Zero-shot Generalization with Subtask Dependencies","date":"2018-07-19","arxiv_id":"1807.07665","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/hierarchical-reinforcement-learning-for-zero#ran","syntology_url":"https://syntology.ai/paper/1807.07665","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.07665"}},"official":{"repos":["srsohn/subtask-graph-execution"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-for-swarm-systems","slug":"deep-reinforcement-learning-for-swarm-systems","title":"Deep Reinforcement Learning for Swarm Systems","date":"2018-07-17","arxiv_id":"1807.06613","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-listen-read-and-follow-score","slug":"learning-to-listen-read-and-follow-score","title":"Learning to Listen, Read, and Follow: Score Following as a Reinforcement Learning Game","date":"2018-07-17","arxiv_id":"1807.06391","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-to-listen-read-and-follow-score#ran","syntology_url":"https://syntology.ai/paper/1807.06391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.06391"}},"official":{"repos":["CPJKU/score_following_game"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/financial-trading-as-a-game-a-deep","slug":"financial-trading-as-a-game-a-deep","title":"Financial Trading as a Game: A Deep Reinforcement Learning Approach","date":"2018-07-08","arxiv_id":"1807.02787","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-continuous","slug":"deep-reinforcement-learning-in-continuous","title":"Deep Reinforcement Learning in Continuous Action Spaces: a Case Study in the Game of Simulated Curling","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-how-to-actively-learn-a-deep","slug":"learning-how-to-actively-learn-a-deep","title":"Learning How to Actively Learn: A Deep Imitation Learning Approach","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sequicity-simplifying-task-oriented-dialogue","slug":"sequicity-simplifying-task-oriented-dialogue","title":"Sequicity: Simplifying Task-oriented Dialogue Systems with Single Sequence-to-Sequence Architectures","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/using-reward-machines-for-high-level-task","slug":"using-reward-machines-for-high-level-task","title":"Using Reward Machines for High-Level Task Specification and Decomposition in Reinforcement Learning","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/illuminating-generalization-in-deep","slug":"illuminating-generalization-in-deep","title":"Illuminating Generalization in Deep Reinforcement Learning through Procedural Level Generation","date":"2018-06-28","arxiv_id":"1806.10729","repositories_listed":1,"syntology":null},{"url":"/paper/qt-opt-scalable-deep-reinforcement-learning","slug":"qt-opt-scalable-deep-reinforcement-learning","title":"QT-Opt: Scalable Deep Reinforcement Learning for Vision-Based Robotic Manipulation","date":"2018-06-27","arxiv_id":"1806.10293","repositories_listed":1,"syntology":null},{"url":"/paper/a-tour-of-reinforcement-learning-the-view","slug":"a-tour-of-reinforcement-learning-the-view","title":"A Tour of Reinforcement Learning: The View from Continuous Control","date":"2018-06-25","arxiv_id":"1806.09460","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-surgical","slug":"deep-reinforcement-learning-for-surgical","title":"Deep Reinforcement Learning for Surgical Gesture Segmentation and Classification","date":"2018-06-21","arxiv_id":"1806.08089","repositories_listed":1,"syntology":null},{"url":"/paper/how-many-random-seeds-statistical-power","slug":"how-many-random-seeds-statistical-power","title":"How Many Random Seeds? Statistical Power Analysis in Deep Reinforcement Learning Experiments","date":"2018-06-21","arxiv_id":"1806.08295","repositories_listed":1,"syntology":null}],"record_sha256":"a0e645231e127359587e51f352896d184bce298dfb99f8b478b63fdda38ed6ee","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}