{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/9","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":132,"rows_per_page":100,"rows":[801,900],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/8","next":"/task/reinforcement-learning/papers/10","papers":[{"url":"/paper/a-hitchhikers-guide-to-statistical","slug":"a-hitchhikers-guide-to-statistical","title":"A Hitchhiker's Guide to Statistical Comparisons of Reinforcement Learning Algorithms","date":"2019-04-15","arxiv_id":"1904.06979","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hitchhikers-guide-to-statistical#ran","syntology_url":"https://syntology.ai/paper/1904.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06979"}},"official":{"repos":["flowersteam/rl_stats","ccolas/rl_stats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/synthesized-policies-for-transfer-and-1","slug":"synthesized-policies-for-transfer-and-1","title":"Synthesized Policies for Transfer and Adaptation across Tasks and Environments","date":"2019-04-05","arxiv_id":"1904.03276","repositories_listed":2,"syntology":null},{"url":"/paper/complexity-weighted-loss-and-diverse","slug":"complexity-weighted-loss-and-diverse","title":"Complexity-Weighted Loss and Diverse Reranking for Sentence Simplification","date":"2019-04-04","arxiv_id":"1904.02767","repositories_listed":2,"syntology":null},{"url":"/paper/meta-learning-acquisition-functions-for","slug":"meta-learning-acquisition-functions-for","title":"Meta-Learning Acquisition Functions for Transfer Learning in Bayesian Optimization","date":"2019-04-04","arxiv_id":"1904.02642","repositories_listed":2,"syntology":null},{"url":"/paper/190408486","slug":"190408486","title":"Meta-learning Convolutional Neural Architectures for Multi-target Concrete Defect Classification with the COncrete DEfect BRidge IMage Dataset","date":"2019-04-02","arxiv_id":"1904.08486","repositories_listed":2,"syntology":null},{"url":"/paper/interpretable-reinforcement-learning-via","slug":"interpretable-reinforcement-learning-via","title":"Optimization Methods for Interpretable Differentiable Decision Trees in Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09338","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interpretable-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/1903.09338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.09338"}},"official":null}},{"url":"/paper/jet-grooming-through-reinforcement-learning","slug":"jet-grooming-through-reinforcement-learning","title":"Jet grooming through reinforcement learning","date":"2019-03-22","arxiv_id":"1903.09644","repositories_listed":2,"syntology":null},{"url":"/paper/batch-policy-learning-under-constraints","slug":"batch-policy-learning-under-constraints","title":"Batch Policy Learning under Constraints","date":"2019-03-20","arxiv_id":"1903.08738","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-feedback","slug":"deep-reinforcement-learning-with-feedback","title":"Deep Reinforcement Learning with Feedback-based Exploration","date":"2019-03-14","arxiv_id":"1903.06151","repositories_listed":2,"syntology":null},{"url":"/paper/ros2learn-a-reinforcement-learning-framework","slug":"ros2learn-a-reinforcement-learning-framework","title":"ROS2Learn: a reinforcement learning framework for ROS 2","date":"2019-03-14","arxiv_id":"1903.06282","repositories_listed":2,"syntology":null},{"url":"/paper/learning-heuristics-over-large-graphs-via","slug":"learning-heuristics-over-large-graphs-via","title":"Learning Heuristics over Large Graphs via Deep Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03332","repositories_listed":2,"syntology":null},{"url":"/paper/skew-fit-state-covering-self-supervised","slug":"skew-fit-state-covering-self-supervised","title":"Skew-Fit: State-Covering Self-Supervised Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03698","repositories_listed":2,"syntology":null},{"url":"/paper/the-streetlearn-environment-and-dataset","slug":"the-streetlearn-environment-and-dataset","title":"The StreetLearn Environment and Dataset","date":"2019-03-04","arxiv_id":"1903.01292","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-streetlearn-environment-and-dataset#ran","syntology_url":"https://syntology.ai/paper/1903.01292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.01292"}},"official":{"repos":["deepmind/streetlearn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-follow-directions-in-street-view","slug":"learning-to-follow-directions-in-street-view","title":"Learning To Follow Directions in Street View","date":"2019-03-01","arxiv_id":"1903.00401","repositories_listed":2,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-for-atari","slug":"model-based-reinforcement-learning-for-atari","title":"Model-Based Reinforcement Learning for Atari","date":"2019-03-01","arxiv_id":"1903.00374","repositories_listed":2,"syntology":{"n":21,"n_ran":16,"n_constructed":8,"n_ran_checked":10,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"16 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-atari#ran","syntology_url":"https://syntology.ai/paper/1903.00374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.00374"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/trojdrl-trojan-attacks-on-deep-reinforcement","slug":"trojdrl-trojan-attacks-on-deep-reinforcement","title":"TrojDRL: Trojan Attacks on Deep Reinforcement Learning Agents","date":"2019-03-01","arxiv_id":"1903.06638","repositories_listed":2,"syntology":null},{"url":"/paper/190504100","slug":"190504100","title":"Deep Reinforcement Learning using Genetic Algorithm for Parameter Optimization","date":"2019-02-19","arxiv_id":"1905.04100","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/190504100#ran","syntology_url":"https://syntology.ai/paper/1905.04100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04100"}},"official":{"repos":["aralab-unr/ReinforcementLearningWithGA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/nutrition-and-health-data-for-cost-sensitive","slug":"nutrition-and-health-data-for-cost-sensitive","title":"Cost-Sensitive Diagnosis and Learning Leveraging Public Health Data","date":"2019-02-19","arxiv_id":"1902.07102","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nutrition-and-health-data-for-cost-sensitive#ran","syntology_url":"https://syntology.ai/paper/1902.07102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.07102"}},"official":{"repos":["mkachuee/Opportunistic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learnable-embedding-space-for-efficient","slug":"learnable-embedding-space-for-efficient","title":"Learnable Embedding Space for Efficient Neural Architecture Compression","date":"2019-02-01","arxiv_id":"1902.00383","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learnable-embedding-space-for-efficient#ran","syntology_url":"https://syntology.ai/paper/1902.00383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00383"}},"official":{"repos":["Friedrich1006/ESNAC"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/visual-hindsight-experience-replay","slug":"visual-hindsight-experience-replay","title":"Addressing Sample Complexity in Visual Tasks Using HER and Hallucinatory GANs","date":"2019-01-31","arxiv_id":"1901.11529","repositories_listed":2,"syntology":null},{"url":"/paper/trust-region-guided-proximal-policy","slug":"trust-region-guided-proximal-policy","title":"Trust Region-Guided Proximal Policy Optimization","date":"2019-01-29","arxiv_id":"1901.10314","repositories_listed":2,"syntology":null},{"url":"/paper/action-robust-reinforcement-learning-and","slug":"action-robust-reinforcement-learning-and","title":"Action Robust Reinforcement Learning and Applications in Continuous Control","date":"2019-01-26","arxiv_id":"1901.09184","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-robust-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/1901.09184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09184"}},"official":{"repos":["tesslerc/ActionRobustRL","icml2019-anonymous-author/Action-Robust-Reinforcement-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-agile-and-dynamic-motor-skills-for","slug":"learning-agile-and-dynamic-motor-skills-for","title":"Learning agile and dynamic motor skills for legged robots","date":"2019-01-24","arxiv_id":"1901.08652","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-agile-and-dynamic-motor-skills-for#ran","syntology_url":"https://syntology.ai/paper/1901.08652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08652"}},"official":{"repos":["junja94/anymal_science_robotics_supplementary"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-multi-agent-reinforcement-learning-in","slug":"the-multi-agent-reinforcement-learning-in","title":"The Multi-Agent Reinforcement Learning in MalmÖ (MARLÖ) Competition","date":"2019-01-23","arxiv_id":"1901.08129","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-multi-agent-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/1901.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08129"}},"official":null}},{"url":"/paper/fast-accurate-and-lightweight-super","slug":"fast-accurate-and-lightweight-super","title":"Fast, Accurate and Lightweight Super-Resolution with Neural Architecture Search","date":"2019-01-22","arxiv_id":"1901.07261","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-accurate-and-lightweight-super#ran","syntology_url":"https://syntology.ai/paper/1901.07261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.07261"}},"official":{"repos":["falsr/FALSR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","repositories_listed":2,"syntology":null},{"url":"/paper/energy-efficient-thermal-comfort-control-in","slug":"energy-efficient-thermal-comfort-control-in","title":"Energy-Efficient Thermal Comfort Control in Smart Buildings via Deep Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04693","repositories_listed":2,"syntology":null},{"url":"/paper/risk-aware-active-inverse-reinforcement","slug":"risk-aware-active-inverse-reinforcement","title":"Risk-Aware Active Inverse Reinforcement Learning","date":"2019-01-08","arxiv_id":"1901.02161","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/risk-aware-active-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.02161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.02161"}},"official":{"repos":["Pearl-UTexas/ActiveVaR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/snas-stochastic-neural-architecture-search","slug":"snas-stochastic-neural-architecture-search","title":"SNAS: Stochastic Neural Architecture Search","date":"2018-12-24","arxiv_id":"1812.09926","repositories_listed":2,"syntology":null},{"url":"/paper/universal-successor-features-approximators","slug":"universal-successor-features-approximators","title":"Universal Successor Features Approximators","date":"2018-12-18","arxiv_id":"1812.07626","repositories_listed":2,"syntology":null},{"url":"/paper/decentralized-computation-offloading-for","slug":"decentralized-computation-offloading-for","title":"Decentralized Computation Offloading for Multi-User Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2018-12-16","arxiv_id":"1812.07394","repositories_listed":2,"syntology":null},{"url":"/paper/knockoff-nets-stealing-functionality-of-black","slug":"knockoff-nets-stealing-functionality-of-black","title":"Knockoff Nets: Stealing Functionality of Black-Box Models","date":"2018-12-06","arxiv_id":"1812.02766","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-learn-how-to-learn-self-adaptive","slug":"learning-to-learn-how-to-learn-self-adaptive","title":"Learning to Learn How to Learn: Self-Adaptive Visual Navigation Using Meta-Learning","date":"2018-12-03","arxiv_id":"1812.00971","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-learn-how-to-learn-self-adaptive#ran","syntology_url":"https://syntology.ai/paper/1812.00971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.00971"}},"official":{"repos":["allenai/savn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-the-softmax-bellman-operator","slug":"revisiting-the-softmax-bellman-operator","title":"Revisiting the Softmax Bellman Operator: New Benefits and New Perspective","date":"2018-12-02","arxiv_id":"1812.00456","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/revisiting-the-softmax-bellman-operator#ran","syntology_url":"https://syntology.ai/paper/1812.00456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.00456"}},"official":{"repos":["zhao-song/Softmax-DQN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/environments-for-lifelong-reinforcement","slug":"environments-for-lifelong-reinforcement","title":"Environments for Lifelong Reinforcement Learning","date":"2018-11-26","arxiv_id":"1811.10732","repositories_listed":2,"syntology":null},{"url":"/paper/learning-goal-embeddings-via-self-play-for","slug":"learning-goal-embeddings-via-self-play-for","title":"Learning Goal Embeddings via Self-Play for Hierarchical Reinforcement Learning","date":"2018-11-22","arxiv_id":"1811.09083","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-and-a-deep","slug":"reinforcement-learning-with-a-and-a-deep","title":"Reinforcement Learning with A* and a Deep Heuristic","date":"2018-11-19","arxiv_id":"1811.07745","repositories_listed":2,"syntology":null},{"url":"/paper/improving-automatic-source-code-summarization","slug":"improving-automatic-source-code-summarization","title":"Improving Automatic Source Code Summarization via Deep Reinforcement Learning","date":"2018-11-17","arxiv_id":"1811.07234","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-automatic-source-code-summarization#ran","syntology_url":"https://syntology.ai/paper/1811.07234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.07234"}},"official":null}},{"url":"/paper/reward-learning-from-human-preferences-and","slug":"reward-learning-from-human-preferences-and","title":"Reward learning from human preferences and demonstrations in Atari","date":"2018-11-15","arxiv_id":"1811.06521","repositories_listed":2,"syntology":null},{"url":"/paper/natural-environment-benchmarks-for","slug":"natural-environment-benchmarks-for","title":"Natural Environment Benchmarks for Reinforcement Learning","date":"2018-11-14","arxiv_id":"1811.06032","repositories_listed":2,"syntology":null},{"url":"/paper/a-hierarchical-framework-for-relation","slug":"a-hierarchical-framework-for-relation","title":"A Hierarchical Framework for Relation Extraction with Reinforcement Learning","date":"2018-11-09","arxiv_id":"1811.03925","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-automatic-test","slug":"reinforcement-learning-for-automatic-test","title":"Reinforcement Learning for Automatic Test Case Prioritization and Selection in Continuous Integration","date":"2018-11-09","arxiv_id":"1811.04122","repositories_listed":2,"syntology":null},{"url":"/paper/memory-based-deep-reinforcement-learning-for","slug":"memory-based-deep-reinforcement-learning-for","title":"Memory-based Deep Reinforcement Learning for Obstacle Avoidance in UAV with Limited Environment Knowledge","date":"2018-11-08","arxiv_id":"1811.03307","repositories_listed":2,"syntology":null},{"url":"/paper/horizon-facebooks-open-source-applied","slug":"horizon-facebooks-open-source-applied","title":"Horizon: Facebook's Open Source Applied Reinforcement Learning Platform","date":"2018-11-01","arxiv_id":"1811.00260","repositories_listed":2,"syntology":null},{"url":"/paper/temporal-regularization-in-markov-decision","slug":"temporal-regularization-in-markov-decision","title":"Temporal Regularization in Markov Decision Process","date":"2018-11-01","arxiv_id":"1811.00429","repositories_listed":2,"syntology":null},{"url":"/paper/differentiable-mpc-for-end-to-end-planning","slug":"differentiable-mpc-for-end-to-end-planning","title":"Differentiable MPC for End-to-end Planning and Control","date":"2018-10-31","arxiv_id":"1810.13400","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/differentiable-mpc-for-end-to-end-planning#ran","syntology_url":"https://syntology.ai/paper/1810.13400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.13400"}},"official":{"repos":["locuslab/differentiable-mpc"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/model-based-active-exploration","slug":"model-based-active-exploration","title":"Model-Based Active Exploration","date":"2018-10-29","arxiv_id":"1810.12162","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-active-exploration#ran","syntology_url":"https://syntology.ai/paper/1810.12162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12162"}},"official":{"repos":["nnaisense/max"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-modular-control-for-embodied-question","slug":"neural-modular-control-for-embodied-question","title":"Neural Modular Control for Embodied Question Answering","date":"2018-10-26","arxiv_id":"1810.11181","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/neural-modular-control-for-embodied-question#ran","syntology_url":"https://syntology.ai/paper/1810.11181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11181"}},"official":null}},{"url":"/paper/making-sense-of-vision-and-touch-self","slug":"making-sense-of-vision-and-touch-self","title":"Making Sense of Vision and Touch: Self-Supervised Learning of Multimodal Representations for Contact-Rich Tasks","date":"2018-10-24","arxiv_id":"1810.10191","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/making-sense-of-vision-and-touch-self#ran","syntology_url":"https://syntology.ai/paper/1810.10191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.10191"}},"official":{"repos":["stanford-iprl-lab/multimodal_representation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-deep-reinforcement-learning-using-online","slug":"fast-deep-reinforcement-learning-using-online","title":"Fast deep reinforcement learning using online adjustments from the past","date":"2018-10-18","arxiv_id":"1810.08163","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-deep-reinforcement-learning-using-online#ran","syntology_url":"https://syntology.ai/paper/1810.08163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.08163"}},"official":null}},{"url":"/paper/successor-uncertainties-exploration-and","slug":"successor-uncertainties-exploration-and","title":"Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","date":"2018-10-15","arxiv_id":"1810.06530","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/successor-uncertainties-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1810.06530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06530"}},"official":null}},{"url":"/paper/q-map-a-convolutional-approach-for-goal","slug":"q-map-a-convolutional-approach-for-goal","title":"Scaling All-Goals Updates in Reinforcement Learning Using Convolutional Neural Networks","date":"2018-10-06","arxiv_id":"1810.02927","repositories_listed":2,"syntology":null},{"url":"/paper/learning-scheduling-algorithms-for-data","slug":"learning-scheduling-algorithms-for-data","title":"Learning Scheduling Algorithms for Data Processing Clusters","date":"2018-10-03","arxiv_id":"1810.01963","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-scheduling-algorithms-for-data#ran","syntology_url":"https://syntology.ai/paper/1810.01963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01963"}},"official":{"repos":["hongzimao/decima-sim"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/cem-rl-combining-evolutionary-and-gradient-1","slug":"cem-rl-combining-evolutionary-and-gradient-1","title":"CEM-RL: Combining evolutionary and gradient-based methods for policy search","date":"2018-10-02","arxiv_id":"1810.01222","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/cem-rl-combining-evolutionary-and-gradient-1#ran","syntology_url":"https://syntology.ai/paper/1810.01222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01222"}},"official":{"repos":["apourchot/CEM-RL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/efficient-dialog-policy-learning-via-positive","slug":"efficient-dialog-policy-learning-via-positive","title":"Efficient Dialog Policy Learning via Positive Memory Retention","date":"2018-10-02","arxiv_id":"1810.01371","repositories_listed":2,"syntology":null},{"url":"/paper/energy-based-hindsight-experience","slug":"energy-based-hindsight-experience","title":"Energy-Based Hindsight Experience Prioritization","date":"2018-10-02","arxiv_id":"1810.01363","repositories_listed":2,"syntology":null},{"url":"/paper/solving-statistical-mechanics-using","slug":"solving-statistical-mechanics-using","title":"Solving Statistical Mechanics Using Variational Autoregressive Networks","date":"2018-09-27","arxiv_id":"1809.10606","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/solving-statistical-mechanics-using#ran","syntology_url":"https://syntology.ai/paper/1809.10606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.10606"}},"official":{"repos":["wangleiphy/VAN.jl","wdphy16/stat-mech-van"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-reinforcement-learning","slug":"benchmarking-reinforcement-learning","title":"Benchmarking Reinforcement Learning Algorithms on Real-World Robots","date":"2018-09-20","arxiv_id":"1809.07731","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1809.07731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.07731"}},"official":{"repos":["kindredresearch/SenseAct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tstarbots-defeating-the-cheating-level","slug":"tstarbots-defeating-the-cheating-level","title":"TStarBots: Defeating the Cheating Level Builtin AI in StarCraft II in the Full Game","date":"2018-09-19","arxiv_id":"1809.07193","repositories_listed":2,"syntology":null},{"url":"/paper/policy-optimization-via-importance-sampling","slug":"policy-optimization-via-importance-sampling","title":"Policy Optimization via Importance Sampling","date":"2018-09-17","arxiv_id":"1809.06098","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-optimization-via-importance-sampling#ran","syntology_url":"https://syntology.ai/paper/1809.06098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06098"}},"official":{"repos":["T3p/pois"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with","slug":"multi-task-deep-reinforcement-learning-with","title":"Multi-task Deep Reinforcement Learning with PopArt","date":"2018-09-12","arxiv_id":"1809.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1809.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.04474"}},"official":null}},{"url":"/paper/apes-a-python-toolbox-for-simulating","slug":"apes-a-python-toolbox-for-simulating","title":"APES: a Python toolbox for simulating reinforcement learning environments","date":"2018-08-31","arxiv_id":"1808.10692","repositories_listed":2,"syntology":null},{"url":"/paper/application-of-self-play-reinforcement","slug":"application-of-self-play-reinforcement","title":"Application of Self-Play Reinforcement Learning to a Four-Player Game of Imperfect Information","date":"2018-08-30","arxiv_id":"1808.10442","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-relation","slug":"reinforcement-learning-for-relation","title":"Reinforcement Learning for Relation Classification from Noisy Data","date":"2018-08-24","arxiv_id":"1808.08013","repositories_listed":2,"syntology":null},{"url":"/paper/towards-machine-learning-based-optimal-has","slug":"towards-machine-learning-based-optimal-has","title":"Towards Machine Learning-Based Optimal HAS","date":"2018-08-24","arxiv_id":"1808.08065","repositories_listed":2,"syntology":null},{"url":"/paper/a-framework-for-automated-cellular-network","slug":"a-framework-for-automated-cellular-network","title":"A Framework for Automated Cellular Network Tuning with Reinforcement Learning","date":"2018-08-13","arxiv_id":"1808.05140","repositories_listed":2,"syntology":null},{"url":"/paper/count-based-exploration-with-the-successor","slug":"count-based-exploration-with-the-successor","title":"Count-Based Exploration with the Successor Representation","date":"2018-07-31","arxiv_id":"1807.11622","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/count-based-exploration-with-the-successor#ran","syntology_url":"https://syntology.ai/paper/1807.11622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.11622"}},"official":{"repos":["mcmachado/count_based_exploration_sr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/remember-and-forget-for-experience-replay","slug":"remember-and-forget-for-experience-replay","title":"Remember and Forget for Experience Replay","date":"2018-07-16","arxiv_id":"1807.05827","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/remember-and-forget-for-experience-replay#ran","syntology_url":"https://syntology.ai/paper/1807.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.05827"}},"official":{"repos":["cselab/smarties"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-reinforcement-learning-with-imagined","slug":"visual-reinforcement-learning-with-imagined","title":"Visual Reinforcement Learning with Imagined Goals","date":"2018-07-12","arxiv_id":"1807.04742","repositories_listed":2,"syntology":null},{"url":"/paper/algorithmic-framework-for-model-based-deep","slug":"algorithmic-framework-for-model-based-deep","title":"Algorithmic Framework for Model-based Deep Reinforcement Learning with Theoretical Guarantees","date":"2018-07-10","arxiv_id":"1807.03858","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/algorithmic-framework-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1807.03858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03858"}},"official":{"repos":["roosephu/slbo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-model-selection-through-adapting-design","slug":"fast-model-selection-through-adapting-design","title":"Optimal design of experiments to identify latent behavioral types","date":"2018-07-10","arxiv_id":"1807.07024","repositories_listed":2,"syntology":null},{"url":"/paper/ranked-reward-enabling-self-play","slug":"ranked-reward-enabling-self-play","title":"Ranked Reward: Enabling Self-Play Reinforcement Learning for Combinatorial Optimization","date":"2018-07-04","arxiv_id":"1807.01672","repositories_listed":2,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with","slug":"sample-efficient-reinforcement-learning-with","title":"Sample-Efficient Reinforcement Learning with Stochastic Ensemble Value Expansion","date":"2018-07-04","arxiv_id":"1807.01675","repositories_listed":2,"syntology":null},{"url":"/paper/policy-optimization-with-penalized-point","slug":"policy-optimization-with-penalized-point","title":"Policy Optimization With Penalized Point Probability Distance: An Alternative To Proximal Policy Optimization","date":"2018-07-02","arxiv_id":"1807.00442","repositories_listed":2,"syntology":null},{"url":"/paper/accuracy-based-curriculum-learning-in-deep","slug":"accuracy-based-curriculum-learning-in-deep","title":"Accuracy-based Curriculum Learning in Deep Reinforcement Learning","date":"2018-06-25","arxiv_id":"1806.09614","repositories_listed":2,"syntology":null},{"url":"/paper/rudder-return-decomposition-for-delayed","slug":"rudder-return-decomposition-for-delayed","title":"RUDDER: Return Decomposition for Delayed Rewards","date":"2018-06-20","arxiv_id":"1806.07857","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rudder-return-decomposition-for-delayed#ran","syntology_url":"https://syntology.ai/paper/1806.07857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07857"}},"official":{"repos":["ml-jku/baselines-rudder","ml-jku/rudder"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-variational-learning-of-bayesian","slug":"structured-variational-learning-of-bayesian","title":"Structured Variational Learning of Bayesian Neural Networks with Horseshoe Priors","date":"2018-06-13","arxiv_id":"1806.05975","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-variational-learning-of-bayesian#ran","syntology_url":"https://syntology.ai/paper/1806.05975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.05975"}},"official":null}},{"url":"/paper/bayesian-model-agnostic-meta-learning","slug":"bayesian-model-agnostic-meta-learning","title":"Bayesian Model-Agnostic Meta-Learning","date":"2018-06-11","arxiv_id":"1806.03836","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bayesian-model-agnostic-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1806.03836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.03836"}},"official":null}},{"url":"/paper/graph-convolutional-policy-network-for-goal","slug":"graph-convolutional-policy-network-for-goal","title":"Graph Convolutional Policy Network for Goal-Directed Molecular Graph Generation","date":"2018-06-07","arxiv_id":"1806.02473","repositories_listed":2,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/graph-convolutional-policy-network-for-goal#ran","syntology_url":"https://syntology.ai/paper/1806.02473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02473"}},"official":{"repos":["bowenliu16/rl_graph_generation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/bayesian-inference-with-anchored-ensembles-of","slug":"bayesian-inference-with-anchored-ensembles-of","title":"Bayesian Inference with Anchored Ensembles of Neural Networks, and Application to Exploration in Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11324","repositories_listed":2,"syntology":null},{"url":"/paper/virtual-taobao-virtualizing-real-world-online","slug":"virtual-taobao-virtualizing-real-world-online","title":"Virtual-Taobao: Virtualizing Real-world Online Retail Environment for Reinforcement Learning","date":"2018-05-25","arxiv_id":"1805.10000","repositories_listed":2,"syntology":null},{"url":"/paper/visceral-machines-reinforcement-learning-with","slug":"visceral-machines-reinforcement-learning-with","title":"Visceral Machines: Risk-Aversion in Reinforcement Learning with Intrinsic Physiological Rewards","date":"2018-05-25","arxiv_id":"1805.09975","repositories_listed":2,"syntology":null},{"url":"/paper/a0c-alpha-zero-in-continuous-action-space","slug":"a0c-alpha-zero-in-continuous-action-space","title":"A0C: Alpha Zero in Continuous Action Space","date":"2018-05-24","arxiv_id":"1805.09613","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a0c-alpha-zero-in-continuous-action-space#ran","syntology_url":"https://syntology.ai/paper/1805.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09613"}},"official":null}},{"url":"/paper/robust-distant-supervision-relation","slug":"robust-distant-supervision-relation","title":"Robust Distant Supervision Relation Extraction via Deep Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09927","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-distant-supervision-relation#ran","syntology_url":"https://syntology.ai/paper/1805.09927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09927"}},"official":null}},{"url":"/paper/verifiable-reinforcement-learning-via-policy","slug":"verifiable-reinforcement-learning-via-policy","title":"Verifiable Reinforcement Learning via Policy Extraction","date":"2018-05-22","arxiv_id":"1805.08328","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-and-control-as","slug":"reinforcement-learning-and-control-as","title":"Reinforcement Learning and Control as Probabilistic Inference: Tutorial and Review","date":"2018-05-02","arxiv_id":"1805.00909","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-and-control-as#ran","syntology_url":"https://syntology.ai/paper/1805.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.00909"}},"official":null}},{"url":"/paper/no-metrics-are-perfect-adversarial-reward","slug":"no-metrics-are-perfect-adversarial-reward","title":"No Metrics Are Perfect: Adversarial Reward Learning for Visual Storytelling","date":"2018-04-24","arxiv_id":"1804.09160","repositories_listed":2,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/no-metrics-are-perfect-adversarial-reward#ran","syntology_url":"https://syntology.ai/paper/1804.09160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.09160"}},"official":{"repos":["littlekobe/AREL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dora-the-explorer-directed-outreaching","slug":"dora-the-explorer-directed-outreaching","title":"DORA The Explorer: Directed Outreaching Reinforcement Action-Selection","date":"2018-04-11","arxiv_id":"1804.04012","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dora-the-explorer-directed-outreaching#ran","syntology_url":"https://syntology.ai/paper/1804.04012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.04012"}},"official":{"repos":["borgr/DORA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/crafting-a-toolchain-for-image-restoration-by","slug":"crafting-a-toolchain-for-image-restoration-by","title":"Crafting a Toolchain for Image Restoration by Deep Reinforcement Learning","date":"2018-04-10","arxiv_id":"1804.03312","repositories_listed":2,"syntology":null},{"url":"/paper/synthesizing-programs-for-images-using","slug":"synthesizing-programs-for-images-using","title":"Synthesizing Programs for Images using Reinforced Adversarial Learning","date":"2018-04-03","arxiv_id":"1804.01118","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-run-challenge-solutions-adapting","slug":"learning-to-run-challenge-solutions-adapting","title":"Learning to Run challenge solutions: Adapting reinforcement learning methods for neuromusculoskeletal environments","date":"2018-04-02","arxiv_id":"1804.00361","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-run-challenge-solutions-adapting#ran","syntology_url":"https://syntology.ai/paper/1804.00361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00361"}},"official":{"repos":["AdamStelmaszczyk/learning2run"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-adapt-in-dynamic-real-world","slug":"learning-to-adapt-in-dynamic-real-world","title":"Learning to Adapt in Dynamic, Real-World Environments Through Meta-Reinforcement Learning","date":"2018-03-30","arxiv_id":"1803.11347","repositories_listed":2,"syntology":null},{"url":"/paper/long-short-term-memory-and-learning-to-learn","slug":"long-short-term-memory-and-learning-to-learn","title":"Long short-term memory and learning-to-learn in networks of spiking neurons","date":"2018-03-26","arxiv_id":"1803.09574","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/long-short-term-memory-and-learning-to-learn#ran","syntology_url":"https://syntology.ai/paper/1803.09574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09574"}},"official":null}},{"url":"/paper/setting-up-a-reinforcement-learning-task-with","slug":"setting-up-a-reinforcement-learning-task-with","title":"Setting up a Reinforcement Learning Task with a Real-World Robot","date":"2018-03-19","arxiv_id":"1803.07067","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-time-series","slug":"deep-reinforcement-learning-for-time-series","title":"Deep reinforcement learning for time series: playing idealized trading games","date":"2018-03-11","arxiv_id":"1803.03916","repositories_listed":2,"syntology":null},{"url":"/paper/variance-networks-when-expectation-does-not","slug":"variance-networks-when-expectation-does-not","title":"Variance Networks: When Expectation Does Not Meet Your Expectations","date":"2018-03-10","arxiv_id":"1803.03764","repositories_listed":2,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/variance-networks-when-expectation-does-not#ran","syntology_url":"https://syntology.ai/paper/1803.03764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.03764"}},"official":{"repos":["da-molchanov/variance-networks"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-predictive-state-policy-networks","slug":"recurrent-predictive-state-policy-networks","title":"Recurrent Predictive State Policy Networks","date":"2018-03-05","arxiv_id":"1803.01489","repositories_listed":2,"syntology":null},{"url":"/paper/accelerating-natural-gradient-with-higher","slug":"accelerating-natural-gradient-with-higher","title":"Accelerating Natural Gradient with Higher-Order Invariance","date":"2018-03-04","arxiv_id":"1803.01273","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accelerating-natural-gradient-with-higher#ran","syntology_url":"https://syntology.ai/paper/1803.01273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.01273"}},"official":{"repos":["ermongroup/higher_order_invariance"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-by-playing-solving-sparse-reward","slug":"learning-by-playing-solving-sparse-reward","title":"Learning by Playing - Solving Sparse Reward Tasks from Scratch","date":"2018-02-28","arxiv_id":"1802.10567","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-playing-solving-sparse-reward#ran","syntology_url":"https://syntology.ai/paper/1802.10567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10567"}},"official":{"repos":["hu-po/pySACQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"6c59cbb2945da4e8df9cff1a4eacf76c6274b6ddc5162914a4bb0a7d3a01a383","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}