{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/8","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":135,"rows_per_page":100,"rows":[701,800],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/7","next":"/task/reinforcement-learning-2/papers/9","papers":[{"url":"/paper/rogue-gym-a-new-challenge-for-generalization","slug":"rogue-gym-a-new-challenge-for-generalization","title":"Rogue-Gym: A New Challenge for Generalization in Reinforcement Learning","date":"2019-04-17","arxiv_id":"1904.08129","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rogue-gym-a-new-challenge-for-generalization#ran","syntology_url":"https://syntology.ai/paper/1904.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08129"}},"official":{"repos":["kngwyu/rogue-gym","kngwyu/rogue-gym-agents-cog19"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hitchhikers-guide-to-statistical","slug":"a-hitchhikers-guide-to-statistical","title":"A Hitchhiker's Guide to Statistical Comparisons of Reinforcement Learning Algorithms","date":"2019-04-15","arxiv_id":"1904.06979","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hitchhikers-guide-to-statistical#ran","syntology_url":"https://syntology.ai/paper/1904.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06979"}},"official":{"repos":["flowersteam/rl_stats","ccolas/rl_stats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-reinforcement-learning-via","slug":"interpretable-reinforcement-learning-via","title":"Optimization Methods for Interpretable Differentiable Decision Trees in Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09338","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interpretable-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/1903.09338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.09338"}},"official":null}},{"url":"/paper/jet-grooming-through-reinforcement-learning","slug":"jet-grooming-through-reinforcement-learning","title":"Jet grooming through reinforcement learning","date":"2019-03-22","arxiv_id":"1903.09644","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-feedback","slug":"deep-reinforcement-learning-with-feedback","title":"Deep Reinforcement Learning with Feedback-based Exploration","date":"2019-03-14","arxiv_id":"1903.06151","repositories_listed":2,"syntology":null},{"url":"/paper/ros2learn-a-reinforcement-learning-framework","slug":"ros2learn-a-reinforcement-learning-framework","title":"ROS2Learn: a reinforcement learning framework for ROS 2","date":"2019-03-14","arxiv_id":"1903.06282","repositories_listed":2,"syntology":null},{"url":"/paper/learning-heuristics-over-large-graphs-via","slug":"learning-heuristics-over-large-graphs-via","title":"Learning Heuristics over Large Graphs via Deep Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03332","repositories_listed":2,"syntology":null},{"url":"/paper/skew-fit-state-covering-self-supervised","slug":"skew-fit-state-covering-self-supervised","title":"Skew-Fit: State-Covering Self-Supervised Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03698","repositories_listed":2,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-for-atari","slug":"model-based-reinforcement-learning-for-atari","title":"Model-Based Reinforcement Learning for Atari","date":"2019-03-01","arxiv_id":"1903.00374","repositories_listed":2,"syntology":{"n":21,"n_ran":16,"n_constructed":8,"n_ran_checked":10,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"16 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-atari#ran","syntology_url":"https://syntology.ai/paper/1903.00374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.00374"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/trojdrl-trojan-attacks-on-deep-reinforcement","slug":"trojdrl-trojan-attacks-on-deep-reinforcement","title":"TrojDRL: Trojan Attacks on Deep Reinforcement Learning Agents","date":"2019-03-01","arxiv_id":"1903.06638","repositories_listed":2,"syntology":null},{"url":"/paper/190504100","slug":"190504100","title":"Deep Reinforcement Learning using Genetic Algorithm for Parameter Optimization","date":"2019-02-19","arxiv_id":"1905.04100","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/190504100#ran","syntology_url":"https://syntology.ai/paper/1905.04100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04100"}},"official":{"repos":["aralab-unr/ReinforcementLearningWithGA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/action-robust-reinforcement-learning-and","slug":"action-robust-reinforcement-learning-and","title":"Action Robust Reinforcement Learning and Applications in Continuous Control","date":"2019-01-26","arxiv_id":"1901.09184","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-robust-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/1901.09184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09184"}},"official":{"repos":["tesslerc/ActionRobustRL","icml2019-anonymous-author/Action-Robust-Reinforcement-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-agile-and-dynamic-motor-skills-for","slug":"learning-agile-and-dynamic-motor-skills-for","title":"Learning agile and dynamic motor skills for legged robots","date":"2019-01-24","arxiv_id":"1901.08652","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-agile-and-dynamic-motor-skills-for#ran","syntology_url":"https://syntology.ai/paper/1901.08652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08652"}},"official":{"repos":["junja94/anymal_science_robotics_supplementary"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-multi-agent-reinforcement-learning-in","slug":"the-multi-agent-reinforcement-learning-in","title":"The Multi-Agent Reinforcement Learning in MalmÖ (MARLÖ) Competition","date":"2019-01-23","arxiv_id":"1901.08129","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-multi-agent-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/1901.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08129"}},"official":null}},{"url":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","repositories_listed":2,"syntology":null},{"url":"/paper/energy-efficient-thermal-comfort-control-in","slug":"energy-efficient-thermal-comfort-control-in","title":"Energy-Efficient Thermal Comfort Control in Smart Buildings via Deep Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04693","repositories_listed":2,"syntology":null},{"url":"/paper/risk-aware-active-inverse-reinforcement","slug":"risk-aware-active-inverse-reinforcement","title":"Risk-Aware Active Inverse Reinforcement Learning","date":"2019-01-08","arxiv_id":"1901.02161","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/risk-aware-active-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.02161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.02161"}},"official":{"repos":["Pearl-UTexas/ActiveVaR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/snas-stochastic-neural-architecture-search","slug":"snas-stochastic-neural-architecture-search","title":"SNAS: Stochastic Neural Architecture Search","date":"2018-12-24","arxiv_id":"1812.09926","repositories_listed":2,"syntology":null},{"url":"/paper/environments-for-lifelong-reinforcement","slug":"environments-for-lifelong-reinforcement","title":"Environments for Lifelong Reinforcement Learning","date":"2018-11-26","arxiv_id":"1811.10732","repositories_listed":2,"syntology":null},{"url":"/paper/learning-goal-embeddings-via-self-play-for","slug":"learning-goal-embeddings-via-self-play-for","title":"Learning Goal Embeddings via Self-Play for Hierarchical Reinforcement Learning","date":"2018-11-22","arxiv_id":"1811.09083","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-and-a-deep","slug":"reinforcement-learning-with-a-and-a-deep","title":"Reinforcement Learning with A* and a Deep Heuristic","date":"2018-11-19","arxiv_id":"1811.07745","repositories_listed":2,"syntology":null},{"url":"/paper/improving-automatic-source-code-summarization","slug":"improving-automatic-source-code-summarization","title":"Improving Automatic Source Code Summarization via Deep Reinforcement Learning","date":"2018-11-17","arxiv_id":"1811.07234","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-automatic-source-code-summarization#ran","syntology_url":"https://syntology.ai/paper/1811.07234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.07234"}},"official":null}},{"url":"/paper/reward-learning-from-human-preferences-and","slug":"reward-learning-from-human-preferences-and","title":"Reward learning from human preferences and demonstrations in Atari","date":"2018-11-15","arxiv_id":"1811.06521","repositories_listed":2,"syntology":null},{"url":"/paper/natural-environment-benchmarks-for","slug":"natural-environment-benchmarks-for","title":"Natural Environment Benchmarks for Reinforcement Learning","date":"2018-11-14","arxiv_id":"1811.06032","repositories_listed":2,"syntology":null},{"url":"/paper/a-hierarchical-framework-for-relation","slug":"a-hierarchical-framework-for-relation","title":"A Hierarchical Framework for Relation Extraction with Reinforcement Learning","date":"2018-11-09","arxiv_id":"1811.03925","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-automatic-test","slug":"reinforcement-learning-for-automatic-test","title":"Reinforcement Learning for Automatic Test Case Prioritization and Selection in Continuous Integration","date":"2018-11-09","arxiv_id":"1811.04122","repositories_listed":2,"syntology":null},{"url":"/paper/horizon-facebooks-open-source-applied","slug":"horizon-facebooks-open-source-applied","title":"Horizon: Facebook's Open Source Applied Reinforcement Learning Platform","date":"2018-11-01","arxiv_id":"1811.00260","repositories_listed":2,"syntology":null},{"url":"/paper/temporal-regularization-in-markov-decision","slug":"temporal-regularization-in-markov-decision","title":"Temporal Regularization in Markov Decision Process","date":"2018-11-01","arxiv_id":"1811.00429","repositories_listed":2,"syntology":null},{"url":"/paper/neural-modular-control-for-embodied-question","slug":"neural-modular-control-for-embodied-question","title":"Neural Modular Control for Embodied Question Answering","date":"2018-10-26","arxiv_id":"1810.11181","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/neural-modular-control-for-embodied-question#ran","syntology_url":"https://syntology.ai/paper/1810.11181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11181"}},"official":null}},{"url":"/paper/fast-deep-reinforcement-learning-using-online","slug":"fast-deep-reinforcement-learning-using-online","title":"Fast deep reinforcement learning using online adjustments from the past","date":"2018-10-18","arxiv_id":"1810.08163","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-deep-reinforcement-learning-using-online#ran","syntology_url":"https://syntology.ai/paper/1810.08163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.08163"}},"official":null}},{"url":"/paper/successor-uncertainties-exploration-and","slug":"successor-uncertainties-exploration-and","title":"Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","date":"2018-10-15","arxiv_id":"1810.06530","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/successor-uncertainties-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1810.06530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06530"}},"official":null}},{"url":"/paper/q-map-a-convolutional-approach-for-goal","slug":"q-map-a-convolutional-approach-for-goal","title":"Scaling All-Goals Updates in Reinforcement Learning Using Convolutional Neural Networks","date":"2018-10-06","arxiv_id":"1810.02927","repositories_listed":2,"syntology":null},{"url":"/paper/energy-based-hindsight-experience","slug":"energy-based-hindsight-experience","title":"Energy-Based Hindsight Experience Prioritization","date":"2018-10-02","arxiv_id":"1810.01363","repositories_listed":2,"syntology":null},{"url":"/paper/solving-statistical-mechanics-using","slug":"solving-statistical-mechanics-using","title":"Solving Statistical Mechanics Using Variational Autoregressive Networks","date":"2018-09-27","arxiv_id":"1809.10606","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/solving-statistical-mechanics-using#ran","syntology_url":"https://syntology.ai/paper/1809.10606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.10606"}},"official":{"repos":["wangleiphy/VAN.jl","wdphy16/stat-mech-van"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-reinforcement-learning","slug":"benchmarking-reinforcement-learning","title":"Benchmarking Reinforcement Learning Algorithms on Real-World Robots","date":"2018-09-20","arxiv_id":"1809.07731","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1809.07731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.07731"}},"official":{"repos":["kindredresearch/SenseAct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with","slug":"multi-task-deep-reinforcement-learning-with","title":"Multi-task Deep Reinforcement Learning with PopArt","date":"2018-09-12","arxiv_id":"1809.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1809.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.04474"}},"official":null}},{"url":"/paper/apes-a-python-toolbox-for-simulating","slug":"apes-a-python-toolbox-for-simulating","title":"APES: a Python toolbox for simulating reinforcement learning environments","date":"2018-08-31","arxiv_id":"1808.10692","repositories_listed":2,"syntology":null},{"url":"/paper/application-of-self-play-reinforcement","slug":"application-of-self-play-reinforcement","title":"Application of Self-Play Reinforcement Learning to a Four-Player Game of Imperfect Information","date":"2018-08-30","arxiv_id":"1808.10442","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-relation","slug":"reinforcement-learning-for-relation","title":"Reinforcement Learning for Relation Classification from Noisy Data","date":"2018-08-24","arxiv_id":"1808.08013","repositories_listed":2,"syntology":null},{"url":"/paper/a-framework-for-automated-cellular-network","slug":"a-framework-for-automated-cellular-network","title":"A Framework for Automated Cellular Network Tuning with Reinforcement Learning","date":"2018-08-13","arxiv_id":"1808.05140","repositories_listed":2,"syntology":null},{"url":"/paper/visual-reinforcement-learning-with-imagined","slug":"visual-reinforcement-learning-with-imagined","title":"Visual Reinforcement Learning with Imagined Goals","date":"2018-07-12","arxiv_id":"1807.04742","repositories_listed":2,"syntology":null},{"url":"/paper/algorithmic-framework-for-model-based-deep","slug":"algorithmic-framework-for-model-based-deep","title":"Algorithmic Framework for Model-based Deep Reinforcement Learning with Theoretical Guarantees","date":"2018-07-10","arxiv_id":"1807.03858","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/algorithmic-framework-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1807.03858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03858"}},"official":{"repos":["roosephu/slbo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ranked-reward-enabling-self-play","slug":"ranked-reward-enabling-self-play","title":"Ranked Reward: Enabling Self-Play Reinforcement Learning for Combinatorial Optimization","date":"2018-07-04","arxiv_id":"1807.01672","repositories_listed":2,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with","slug":"sample-efficient-reinforcement-learning-with","title":"Sample-Efficient Reinforcement Learning with Stochastic Ensemble Value Expansion","date":"2018-07-04","arxiv_id":"1807.01675","repositories_listed":2,"syntology":null},{"url":"/paper/accuracy-based-curriculum-learning-in-deep","slug":"accuracy-based-curriculum-learning-in-deep","title":"Accuracy-based Curriculum Learning in Deep Reinforcement Learning","date":"2018-06-25","arxiv_id":"1806.09614","repositories_listed":2,"syntology":null},{"url":"/paper/rudder-return-decomposition-for-delayed","slug":"rudder-return-decomposition-for-delayed","title":"RUDDER: Return Decomposition for Delayed Rewards","date":"2018-06-20","arxiv_id":"1806.07857","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rudder-return-decomposition-for-delayed#ran","syntology_url":"https://syntology.ai/paper/1806.07857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07857"}},"official":{"repos":["ml-jku/baselines-rudder","ml-jku/rudder"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-variational-learning-of-bayesian","slug":"structured-variational-learning-of-bayesian","title":"Structured Variational Learning of Bayesian Neural Networks with Horseshoe Priors","date":"2018-06-13","arxiv_id":"1806.05975","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-variational-learning-of-bayesian#ran","syntology_url":"https://syntology.ai/paper/1806.05975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.05975"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/bayesian-inference-with-anchored-ensembles-of","slug":"bayesian-inference-with-anchored-ensembles-of","title":"Bayesian Inference with Anchored Ensembles of Neural Networks, and Application to Exploration in Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11324","repositories_listed":2,"syntology":null},{"url":"/paper/virtual-taobao-virtualizing-real-world-online","slug":"virtual-taobao-virtualizing-real-world-online","title":"Virtual-Taobao: Virtualizing Real-world Online Retail Environment for Reinforcement Learning","date":"2018-05-25","arxiv_id":"1805.10000","repositories_listed":2,"syntology":null},{"url":"/paper/visceral-machines-reinforcement-learning-with","slug":"visceral-machines-reinforcement-learning-with","title":"Visceral Machines: Risk-Aversion in Reinforcement Learning with Intrinsic Physiological Rewards","date":"2018-05-25","arxiv_id":"1805.09975","repositories_listed":2,"syntology":null},{"url":"/paper/a0c-alpha-zero-in-continuous-action-space","slug":"a0c-alpha-zero-in-continuous-action-space","title":"A0C: Alpha Zero in Continuous Action Space","date":"2018-05-24","arxiv_id":"1805.09613","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a0c-alpha-zero-in-continuous-action-space#ran","syntology_url":"https://syntology.ai/paper/1805.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09613"}},"official":null}},{"url":"/paper/robust-distant-supervision-relation","slug":"robust-distant-supervision-relation","title":"Robust Distant Supervision Relation Extraction via Deep Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09927","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-distant-supervision-relation#ran","syntology_url":"https://syntology.ai/paper/1805.09927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09927"}},"official":null}},{"url":"/paper/verifiable-reinforcement-learning-via-policy","slug":"verifiable-reinforcement-learning-via-policy","title":"Verifiable Reinforcement Learning via Policy Extraction","date":"2018-05-22","arxiv_id":"1805.08328","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-and-control-as","slug":"reinforcement-learning-and-control-as","title":"Reinforcement Learning and Control as Probabilistic Inference: Tutorial and Review","date":"2018-05-02","arxiv_id":"1805.00909","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-and-control-as#ran","syntology_url":"https://syntology.ai/paper/1805.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.00909"}},"official":null}},{"url":"/paper/crafting-a-toolchain-for-image-restoration-by","slug":"crafting-a-toolchain-for-image-restoration-by","title":"Crafting a Toolchain for Image Restoration by Deep Reinforcement Learning","date":"2018-04-10","arxiv_id":"1804.03312","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-run-challenge-solutions-adapting","slug":"learning-to-run-challenge-solutions-adapting","title":"Learning to Run challenge solutions: Adapting reinforcement learning methods for neuromusculoskeletal environments","date":"2018-04-02","arxiv_id":"1804.00361","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-run-challenge-solutions-adapting#ran","syntology_url":"https://syntology.ai/paper/1804.00361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00361"}},"official":{"repos":["AdamStelmaszczyk/learning2run"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-adapt-in-dynamic-real-world","slug":"learning-to-adapt-in-dynamic-real-world","title":"Learning to Adapt in Dynamic, Real-World Environments Through Meta-Reinforcement Learning","date":"2018-03-30","arxiv_id":"1803.11347","repositories_listed":2,"syntology":null},{"url":"/paper/setting-up-a-reinforcement-learning-task-with","slug":"setting-up-a-reinforcement-learning-task-with","title":"Setting up a Reinforcement Learning Task with a Real-World Robot","date":"2018-03-19","arxiv_id":"1803.07067","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-time-series","slug":"deep-reinforcement-learning-for-time-series","title":"Deep reinforcement learning for time series: playing idealized trading games","date":"2018-03-11","arxiv_id":"1803.03916","repositories_listed":2,"syntology":null},{"url":"/paper/learning-by-playing-solving-sparse-reward","slug":"learning-by-playing-solving-sparse-reward","title":"Learning by Playing - Solving Sparse Reward Tasks from Scratch","date":"2018-02-28","arxiv_id":"1802.10567","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-playing-solving-sparse-reward#ran","syntology_url":"https://syntology.ai/paper/1802.10567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10567"}},"official":{"repos":["hu-po/pySACQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/model-ensemble-trust-region-policy","slug":"model-ensemble-trust-region-policy","title":"Model-Ensemble Trust-Region Policy Optimization","date":"2018-02-28","arxiv_id":"1802.10592","repositories_listed":2,"syntology":{"n":6,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 6 unverified","sample_list":"/paper/model-ensemble-trust-region-policy#ran","syntology_url":"https://syntology.ai/paper/1802.10592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10592"}},"official":{"repos":["thanard/me-trpo"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":[]}}},{"url":"/paper/meta-reinforcement-learning-of-structured","slug":"meta-reinforcement-learning-of-structured","title":"Meta-Reinforcement Learning of Structured Exploration Strategies","date":"2018-02-20","arxiv_id":"1802.07245","repositories_listed":2,"syntology":null},{"url":"/paper/monte-carlo-q-learning-for-general-game","slug":"monte-carlo-q-learning-for-general-game","title":"Monte Carlo Q-learning for General Game Playing","date":"2018-02-16","arxiv_id":"1802.05944","repositories_listed":2,"syntology":null},{"url":"/paper/multimodal-sentiment-analysis-with-word-level","slug":"multimodal-sentiment-analysis-with-word-level","title":"Multimodal Sentiment Analysis with Word-Level Fusion and Reinforcement Learning","date":"2018-02-03","arxiv_id":"1802.00924","repositories_listed":2,"syntology":null},{"url":"/paper/using-reinforcement-learning-to-learn-how-to","slug":"using-reinforcement-learning-to-learn-how-to","title":"Using reinforcement learning to learn how to play text-based games","date":"2018-01-06","arxiv_id":"1801.01999","repositories_listed":2,"syntology":null},{"url":"/paper/multi-timescale-memory-dynamics-in-a","slug":"multi-timescale-memory-dynamics-in-a","title":"Multi-timescale memory dynamics in a reinforcement learning network with attention-gated memory","date":"2017-12-28","arxiv_id":"1712.10062","repositories_listed":2,"syntology":null},{"url":"/paper/improving-exploration-in-evolution-strategies","slug":"improving-exploration-in-evolution-strategies","title":"Improving Exploration in Evolution Strategies for Deep Reinforcement Learning via a Population of Novelty-Seeking Agents","date":"2017-12-18","arxiv_id":"1712.06560","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-exploration-in-evolution-strategies#ran","syntology_url":"https://syntology.ai/paper/1712.06560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.06560"}},"official":{"repos":["uber-research/deep-neuroevolution"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/ai2-thor-an-interactive-3d-environment-for","slug":"ai2-thor-an-interactive-3d-environment-for","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","date":"2017-12-14","arxiv_id":"1712.05474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai2-thor-an-interactive-3d-environment-for#ran","syntology_url":"https://syntology.ai/paper/1712.05474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.05474"}},"official":{"repos":["allenai/ai2thor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/differentiable-lower-bound-for-expected-bleu","slug":"differentiable-lower-bound-for-expected-bleu","title":"Differentiable lower bound for expected BLEU score","date":"2017-12-13","arxiv_id":"1712.04708","repositories_listed":2,"syntology":null},{"url":"/paper/minos-multimodal-indoor-simulator-for","slug":"minos-multimodal-indoor-simulator-for","title":"MINOS: Multimodal Indoor Simulator for Navigation in Complex Environments","date":"2017-12-11","arxiv_id":"1712.03931","repositories_listed":2,"syntology":null},{"url":"/paper/ai-safety-gridworlds","slug":"ai-safety-gridworlds","title":"AI Safety Gridworlds","date":"2017-11-27","arxiv_id":"1711.09883","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sepsis","slug":"deep-reinforcement-learning-for-sepsis","title":"Deep Reinforcement Learning for Sepsis Treatment","date":"2017-11-27","arxiv_id":"1711.09602","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sepsis#ran","syntology_url":"https://syntology.ai/paper/1711.09602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.09602"}},"official":{"repos":["darkefyre/sepsisrl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/action-depedent-control-variates-for-policy","slug":"action-depedent-control-variates-for-policy","title":"Action-depedent Control Variates for Policy Optimization via Stein's Identity","date":"2017-10-30","arxiv_id":"1710.11198","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/action-depedent-control-variates-for-policy#ran","syntology_url":"https://syntology.ai/paper/1710.11198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11198"}},"official":null}},{"url":"/paper/detecting-adversarial-attacks-on-neural","slug":"detecting-adversarial-attacks-on-neural","title":"Detecting Adversarial Attacks on Neural Network Policies with Visual Foresight","date":"2017-10-02","arxiv_id":"1710.00814","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/detecting-adversarial-attacks-on-neural#ran","syntology_url":"https://syntology.ai/paper/1710.00814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.00814"}},"official":{"repos":["yenchenlin/rl-attack-detection"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-deep-reinforcement-learning","slug":"self-supervised-deep-reinforcement-learning","title":"Self-supervised Deep Reinforcement Learning with Generalized Computation Graphs for Robot Navigation","date":"2017-09-29","arxiv_id":"1709.10489","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1709.10489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.10489"}},"official":{"repos":["gkahn13/gcg"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-tamer-interactive-agent-shaping-in-high","slug":"deep-tamer-interactive-agent-shaping-in-high","title":"Deep TAMER: Interactive Agent Shaping in High-Dimensional State Spaces","date":"2017-09-28","arxiv_id":"1709.10163","repositories_listed":2,"syntology":null},{"url":"/paper/towards-optimally-decentralized-multi-robot","slug":"towards-optimally-decentralized-multi-robot","title":"Towards Optimally Decentralized Multi-Robot Collision Avoidance via Deep Reinforcement Learning","date":"2017-09-28","arxiv_id":"1709.10082","repositories_listed":2,"syntology":null},{"url":"/paper/neural-optimizer-search-with-reinforcement","slug":"neural-optimizer-search-with-reinforcement","title":"Neural Optimizer Search with Reinforcement Learning","date":"2017-09-21","arxiv_id":"1709.07417","repositories_listed":2,"syntology":null},{"url":"/paper/tensorflow-agents-efficient-batched","slug":"tensorflow-agents-efficient-batched","title":"TensorFlow Agents: Efficient Batched Reinforcement Learning in TensorFlow","date":"2017-09-08","arxiv_id":"1709.02878","repositories_listed":2,"syntology":null},{"url":"/paper/mean-actor-critic","slug":"mean-actor-critic","title":"Mean Actor Critic","date":"2017-09-01","arxiv_id":"1709.00503","repositories_listed":2,"syntology":null},{"url":"/paper/deeppath-a-reinforcement-learning-method-for","slug":"deeppath-a-reinforcement-learning-method-for","title":"DeepPath: A Reinforcement Learning Method for Knowledge Graph Reasoning","date":"2017-07-20","arxiv_id":"1707.06690","repositories_listed":2,"syntology":null},{"url":"/paper/imagination-augmented-agents-for-deep","slug":"imagination-augmented-agents-for-deep","title":"Imagination-Augmented Agents for Deep Reinforcement Learning","date":"2017-07-19","arxiv_id":"1707.06203","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imagination-augmented-agents-for-deep#ran","syntology_url":"https://syntology.ai/paper/1707.06203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.06203"}},"official":null}},{"url":"/paper/elf-an-extensive-lightweight-and-flexible","slug":"elf-an-extensive-lightweight-and-flexible","title":"ELF: An Extensive, Lightweight and Flexible Research Platform for Real-time Strategy Games","date":"2017-07-04","arxiv_id":"1707.01067","repositories_listed":2,"syntology":null},{"url":"/paper/ask-the-right-questions-active-question","slug":"ask-the-right-questions-active-question","title":"Ask the Right Questions: Active Question Reformulation with Reinforcement Learning","date":"2017-05-22","arxiv_id":"1705.07830","repositories_listed":2,"syntology":null},{"url":"/paper/task-oriented-query-reformulation-with-1","slug":"task-oriented-query-reformulation-with-1","title":"Task-Oriented Query Reformulation with Reinforcement Learning","date":"2017-04-15","arxiv_id":"1704.04572","repositories_listed":2,"syntology":null},{"url":"/paper/stochastic-neural-networks-for-hierarchical","slug":"stochastic-neural-networks-for-hierarchical","title":"Stochastic Neural Networks for Hierarchical Reinforcement Learning","date":"2017-04-10","arxiv_id":"1704.03012","repositories_listed":2,"syntology":null},{"url":"/paper/learning-visual-servoing-with-deep-features","slug":"learning-visual-servoing-with-deep-features","title":"Learning Visual Servoing with Deep Features and Fitted Q-Iteration","date":"2017-03-31","arxiv_id":"1703.11000","repositories_listed":2,"syntology":null},{"url":"/paper/dynamic-computational-time-for-visual","slug":"dynamic-computational-time-for-visual","title":"Dynamic Computational Time for Visual Attention","date":"2017-03-30","arxiv_id":"1703.10332","repositories_listed":2,"syntology":null},{"url":"/paper/socially-aware-motion-planning-with-deep","slug":"socially-aware-motion-planning-with-deep","title":"Socially Aware Motion Planning with Deep Reinforcement Learning","date":"2017-03-26","arxiv_id":"1703.08862","repositories_listed":2,"syntology":null},{"url":"/paper/autonomous-braking-system-via-deep","slug":"autonomous-braking-system-via-deep","title":"Autonomous Braking System via Deep Reinforcement Learning","date":"2017-02-08","arxiv_id":"1702.02302","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-an-overview","slug":"deep-reinforcement-learning-an-overview","title":"Deep Reinforcement Learning: An Overview","date":"2017-01-25","arxiv_id":"1701.07274","repositories_listed":2,"syntology":null},{"url":"/paper/learning-light-transport-the-reinforced-way","slug":"learning-light-transport-the-reinforced-way","title":"Learning Light Transport the Reinforced Way","date":"2017-01-25","arxiv_id":"1701.07403","repositories_listed":2,"syntology":null},{"url":"/paper/learning-through-dialogue-interactions-by","slug":"learning-through-dialogue-interactions-by","title":"Learning through Dialogue Interactions by Asking Questions","date":"2016-12-15","arxiv_id":"1612.04936","repositories_listed":2,"syntology":null},{"url":"/paper/dialogue-learning-with-human-in-the-loop","slug":"dialogue-learning-with-human-in-the-loop","title":"Dialogue Learning With Human-In-The-Loop","date":"2016-11-29","arxiv_id":"1611.09823","repositories_listed":2,"syntology":null},{"url":"/paper/modular-multitask-reinforcement-learning-with","slug":"modular-multitask-reinforcement-learning-with","title":"Modular Multitask Reinforcement Learning with Policy Sketches","date":"2016-11-06","arxiv_id":"1611.01796","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modular-multitask-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1611.01796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01796"}},"official":{"repos":["jacobandreas/psketch"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-objective-deep-reinforcement-learning","slug":"multi-objective-deep-reinforcement-learning","title":"Multi-Objective Deep Reinforcement Learning","date":"2016-10-09","arxiv_id":"1610.02707","repositories_listed":2,"syntology":null},{"url":"/paper/target-driven-visual-navigation-in-indoor","slug":"target-driven-visual-navigation-in-indoor","title":"Target-driven Visual Navigation in Indoor Scenes using Deep Reinforcement Learning","date":"2016-09-16","arxiv_id":"1609.05143","repositories_listed":2,"syntology":null},{"url":"/paper/a-greedy-approach-to-adapting-the-trace","slug":"a-greedy-approach-to-adapting-the-trace","title":"A Greedy Approach to Adapting the Trace Parameter for Temporal Difference Learning","date":"2016-07-02","arxiv_id":"1607.00446","repositories_listed":2,"syntology":null},{"url":"/paper/cooperative-inverse-reinforcement-learning","slug":"cooperative-inverse-reinforcement-learning","title":"Cooperative Inverse Reinforcement Learning","date":"2016-06-09","arxiv_id":"1606.03137","repositories_listed":2,"syntology":null}],"record_sha256":"0db6efead133db323267c7635c657e5e2b24d5e460a5e49bf75e5cc2103afc09","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}