{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/9","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":152,"rows_per_page":100,"rows":[801,900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/8","next":"/task/reinforcement-learning-1/papers/10","papers":[{"url":"/paper/deep-active-inference-as-variational-policy","slug":"deep-active-inference-as-variational-policy","title":"Deep Active Inference as Variational Policy Gradients","date":"2019-07-08","arxiv_id":"1907.03876","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-model-based-reinforcement","slug":"benchmarking-model-based-reinforcement","title":"Benchmarking Model-Based Reinforcement Learning","date":"2019-07-03","arxiv_id":"1907.02057","repositories_listed":2,"syntology":null},{"url":"/paper/playing-adaptively-against-stealthy-opponents","slug":"playing-adaptively-against-stealthy-opponents","title":"QFlip: An Adaptive Reinforcement Learning Strategy for the FlipIt Security Game","date":"2019-06-27","arxiv_id":"1906.11938","repositories_listed":2,"syntology":null},{"url":"/paper/modern-deep-reinforcement-learning-algorithms","slug":"modern-deep-reinforcement-learning-algorithms","title":"Modern Deep Reinforcement Learning Algorithms","date":"2019-06-24","arxiv_id":"1906.10025","repositories_listed":2,"syntology":null},{"url":"/paper/language-as-an-abstraction-for-hierarchical","slug":"language-as-an-abstraction-for-hierarchical","title":"Language as an Abstraction for Hierarchical Deep Reinforcement Learning","date":"2019-06-18","arxiv_id":"1906.07343","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-as-an-abstraction-for-hierarchical#ran","syntology_url":"https://syntology.ai/paper/1906.07343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.07343"}},"official":{"repos":["google-research/clevr_robot_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/moet-interpretable-and-verifiable","slug":"moet-interpretable-and-verifiable","title":"MoËT: Mixture of Expert Trees and its Application to Verifiable Reinforcement Learning","date":"2019-06-16","arxiv_id":"1906.06717","repositories_listed":2,"syntology":null},{"url":"/paper/cross-view-policy-learning-for-street","slug":"cross-view-policy-learning-for-street","title":"Cross-View Policy Learning for Street Navigation","date":"2019-06-13","arxiv_id":"1906.05930","repositories_listed":2,"syntology":null},{"url":"/paper/when-to-use-parametric-models-in","slug":"when-to-use-parametric-models-in","title":"When to use parametric models in reinforcement learning?","date":"2019-06-12","arxiv_id":"1906.05243","repositories_listed":2,"syntology":null},{"url":"/paper/intrinsically-efficient-stable-and-bounded","slug":"intrinsically-efficient-stable-and-bounded","title":"Intrinsically Efficient, Stable, and Bounded Off-Policy Evaluation for Reinforcement Learning","date":"2019-06-09","arxiv_id":"1906.03735","repositories_listed":2,"syntology":null},{"url":"/paper/190600572","slug":"190600572","title":"Using a Logarithmic Mapping to Enable Lower Discount Factors in Reinforcement Learning","date":"2019-06-03","arxiv_id":"1906.00572","repositories_listed":2,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-through-a","slug":"explainable-reinforcement-learning-through-a","title":"Explainable Reinforcement Learning Through a Causal Lens","date":"2019-05-27","arxiv_id":"1905.10958","repositories_listed":2,"syntology":null},{"url":"/paper/interactive-differentiable-simulation","slug":"interactive-differentiable-simulation","title":"Interactive Differentiable Simulation","date":"2019-05-26","arxiv_id":"1905.10706","repositories_listed":2,"syntology":null},{"url":"/paper/adversarial-policies-attacking-deep","slug":"adversarial-policies-attacking-deep","title":"Adversarial Policies: Attacking Deep Reinforcement Learning","date":"2019-05-25","arxiv_id":"1905.10615","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adversarial-policies-attacking-deep#ran","syntology_url":"https://syntology.ai/paper/1905.10615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.10615"}},"official":{"repos":["HumanCompatibleAI/adversarial-policies"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-dual-reinforcement-learning-framework-for","slug":"a-dual-reinforcement-learning-framework-for","title":"A Dual Reinforcement Learning Framework for Unsupervised Text Style Transfer","date":"2019-05-24","arxiv_id":"1905.10060","repositories_listed":2,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/a-dual-reinforcement-learning-framework-for#ran","syntology_url":"https://syntology.ai/paper/1905.10060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.10060"}},"official":{"repos":["luofuli/DualLanST"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/continual-reinforcement-learning-in-3d-non","slug":"continual-reinforcement-learning-in-3d-non","title":"Continual Reinforcement Learning in 3D Non-stationary Environments","date":"2019-05-24","arxiv_id":"1905.10112","repositories_listed":2,"syntology":null},{"url":"/paper/estimating-risk-and-uncertainty-in-deep","slug":"estimating-risk-and-uncertainty-in-deep","title":"Estimating Risk and Uncertainty in Deep Reinforcement Learning","date":"2019-05-23","arxiv_id":"1905.09638","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/estimating-risk-and-uncertainty-in-deep#ran","syntology_url":"https://syntology.ai/paper/1905.09638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09638"}},"official":{"repos":["IndustAI/risk-and-uncertainty"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-reinforcement-learning-in-contextual","slug":"inverse-reinforcement-learning-in-contextual","title":"Inverse Reinforcement Learning in Contextual MDPs","date":"2019-05-23","arxiv_id":"1905.09710","repositories_listed":2,"syntology":null},{"url":"/paper/cobra-data-efficient-model-based-rl-through","slug":"cobra-data-efficient-model-based-rl-through","title":"COBRA: Data-Efficient Model-Based RL through Unsupervised Object Discovery and Curiosity-Driven Exploration","date":"2019-05-22","arxiv_id":"1905.09275","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cobra-data-efficient-model-based-rl-through#ran","syntology_url":"https://syntology.ai/paper/1905.09275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09275"}},"official":{"repos":["deepmind/spriteworld"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-reinforcement-learning-with-task","slug":"meta-reinforcement-learning-with-task","title":"Meta Reinforcement Learning with Task Embedding and Shared Policy","date":"2019-05-16","arxiv_id":"1905.06527","repositories_listed":2,"syntology":null},{"url":"/paper/random-expert-distillation-imitation-learning","slug":"random-expert-distillation-imitation-learning","title":"Random Expert Distillation: Imitation Learning via Expert Policy Support Estimation","date":"2019-05-16","arxiv_id":"1905.06750","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/random-expert-distillation-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1905.06750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.06750"}},"official":{"repos":["RuohanW/RED"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-based-off-policy-deep","slug":"trajectory-based-off-policy-deep","title":"Trajectory-Based Off-Policy Deep Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05710","repositories_listed":2,"syntology":null},{"url":"/paper/toybox-a-suite-of-environments-for","slug":"toybox-a-suite-of-environments-for","title":"Toybox: A Suite of Environments for Experimental Evaluation of Deep Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.02825","repositories_listed":2,"syntology":null},{"url":"/paper/rl-gan-net-a-reinforcement-learning-agent","slug":"rl-gan-net-a-reinforcement-learning-agent","title":"RL-GAN-Net: A Reinforcement Learning Agent Controlled GAN Network for Real-Time Point Cloud Shape Completion","date":"2019-04-28","arxiv_id":"1904.12304","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rl-gan-net-a-reinforcement-learning-agent#ran","syntology_url":"https://syntology.ai/paper/1904.12304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.12304"}},"official":null}},{"url":"/paper/safe-reinforcement-learning-with-scene","slug":"safe-reinforcement-learning-with-scene","title":"Safe Reinforcement Learning with Scene Decomposition for Navigating Complex Urban Environments","date":"2019-04-25","arxiv_id":"1904.11483","repositories_listed":2,"syntology":null},{"url":"/paper/baconian-a-unified-opensource-framework-for","slug":"baconian-a-unified-opensource-framework-for","title":"Baconian: A Unified Open-source Framework for Model-Based Reinforcement Learning","date":"2019-04-23","arxiv_id":"1904.10762","repositories_listed":2,"syntology":null},{"url":"/paper/model-free-deep-reinforcement-learning-for","slug":"model-free-deep-reinforcement-learning-for","title":"Model-free Deep Reinforcement Learning for Urban Autonomous Driving","date":"2019-04-20","arxiv_id":"1904.09503","repositories_listed":2,"syntology":null},{"url":"/paper/rogue-gym-a-new-challenge-for-generalization","slug":"rogue-gym-a-new-challenge-for-generalization","title":"Rogue-Gym: A New Challenge for Generalization in Reinforcement Learning","date":"2019-04-17","arxiv_id":"1904.08129","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rogue-gym-a-new-challenge-for-generalization#ran","syntology_url":"https://syntology.ai/paper/1904.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08129"}},"official":{"repos":["kngwyu/rogue-gym","kngwyu/rogue-gym-agents-cog19"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hitchhikers-guide-to-statistical","slug":"a-hitchhikers-guide-to-statistical","title":"A Hitchhiker's Guide to Statistical Comparisons of Reinforcement Learning Algorithms","date":"2019-04-15","arxiv_id":"1904.06979","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hitchhikers-guide-to-statistical#ran","syntology_url":"https://syntology.ai/paper/1904.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06979"}},"official":{"repos":["flowersteam/rl_stats","ccolas/rl_stats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/190408486","slug":"190408486","title":"Meta-learning Convolutional Neural Architectures for Multi-target Concrete Defect Classification with the COncrete DEfect BRidge IMage Dataset","date":"2019-04-02","arxiv_id":"1904.08486","repositories_listed":2,"syntology":null},{"url":"/paper/interpretable-reinforcement-learning-via","slug":"interpretable-reinforcement-learning-via","title":"Optimization Methods for Interpretable Differentiable Decision Trees in Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09338","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interpretable-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/1903.09338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.09338"}},"official":null}},{"url":"/paper/jet-grooming-through-reinforcement-learning","slug":"jet-grooming-through-reinforcement-learning","title":"Jet grooming through reinforcement learning","date":"2019-03-22","arxiv_id":"1903.09644","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-feedback","slug":"deep-reinforcement-learning-with-feedback","title":"Deep Reinforcement Learning with Feedback-based Exploration","date":"2019-03-14","arxiv_id":"1903.06151","repositories_listed":2,"syntology":null},{"url":"/paper/ros2learn-a-reinforcement-learning-framework","slug":"ros2learn-a-reinforcement-learning-framework","title":"ROS2Learn: a reinforcement learning framework for ROS 2","date":"2019-03-14","arxiv_id":"1903.06282","repositories_listed":2,"syntology":null},{"url":"/paper/learning-heuristics-over-large-graphs-via","slug":"learning-heuristics-over-large-graphs-via","title":"Learning Heuristics over Large Graphs via Deep Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03332","repositories_listed":2,"syntology":null},{"url":"/paper/skew-fit-state-covering-self-supervised","slug":"skew-fit-state-covering-self-supervised","title":"Skew-Fit: State-Covering Self-Supervised Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03698","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-follow-directions-in-street-view","slug":"learning-to-follow-directions-in-street-view","title":"Learning To Follow Directions in Street View","date":"2019-03-01","arxiv_id":"1903.00401","repositories_listed":2,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-for-atari","slug":"model-based-reinforcement-learning-for-atari","title":"Model-Based Reinforcement Learning for Atari","date":"2019-03-01","arxiv_id":"1903.00374","repositories_listed":2,"syntology":{"n":21,"n_ran":16,"n_constructed":8,"n_ran_checked":10,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"16 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-atari#ran","syntology_url":"https://syntology.ai/paper/1903.00374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.00374"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/trojdrl-trojan-attacks-on-deep-reinforcement","slug":"trojdrl-trojan-attacks-on-deep-reinforcement","title":"TrojDRL: Trojan Attacks on Deep Reinforcement Learning Agents","date":"2019-03-01","arxiv_id":"1903.06638","repositories_listed":2,"syntology":null},{"url":"/paper/190504100","slug":"190504100","title":"Deep Reinforcement Learning using Genetic Algorithm for Parameter Optimization","date":"2019-02-19","arxiv_id":"1905.04100","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/190504100#ran","syntology_url":"https://syntology.ai/paper/1905.04100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04100"}},"official":{"repos":["aralab-unr/ReinforcementLearningWithGA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/visual-hindsight-experience-replay","slug":"visual-hindsight-experience-replay","title":"Addressing Sample Complexity in Visual Tasks Using HER and Hallucinatory GANs","date":"2019-01-31","arxiv_id":"1901.11529","repositories_listed":2,"syntology":null},{"url":"/paper/trust-region-guided-proximal-policy","slug":"trust-region-guided-proximal-policy","title":"Trust Region-Guided Proximal Policy Optimization","date":"2019-01-29","arxiv_id":"1901.10314","repositories_listed":2,"syntology":null},{"url":"/paper/action-robust-reinforcement-learning-and","slug":"action-robust-reinforcement-learning-and","title":"Action Robust Reinforcement Learning and Applications in Continuous Control","date":"2019-01-26","arxiv_id":"1901.09184","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-robust-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/1901.09184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09184"}},"official":{"repos":["tesslerc/ActionRobustRL","icml2019-anonymous-author/Action-Robust-Reinforcement-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-agile-and-dynamic-motor-skills-for","slug":"learning-agile-and-dynamic-motor-skills-for","title":"Learning agile and dynamic motor skills for legged robots","date":"2019-01-24","arxiv_id":"1901.08652","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-agile-and-dynamic-motor-skills-for#ran","syntology_url":"https://syntology.ai/paper/1901.08652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08652"}},"official":{"repos":["junja94/anymal_science_robotics_supplementary"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-multi-agent-reinforcement-learning-in","slug":"the-multi-agent-reinforcement-learning-in","title":"The Multi-Agent Reinforcement Learning in MalmÖ (MARLÖ) Competition","date":"2019-01-23","arxiv_id":"1901.08129","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-multi-agent-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/1901.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08129"}},"official":null}},{"url":"/paper/fast-accurate-and-lightweight-super","slug":"fast-accurate-and-lightweight-super","title":"Fast, Accurate and Lightweight Super-Resolution with Neural Architecture Search","date":"2019-01-22","arxiv_id":"1901.07261","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-accurate-and-lightweight-super#ran","syntology_url":"https://syntology.ai/paper/1901.07261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.07261"}},"official":{"repos":["falsr/FALSR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","repositories_listed":2,"syntology":null},{"url":"/paper/energy-efficient-thermal-comfort-control-in","slug":"energy-efficient-thermal-comfort-control-in","title":"Energy-Efficient Thermal Comfort Control in Smart Buildings via Deep Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04693","repositories_listed":2,"syntology":null},{"url":"/paper/risk-aware-active-inverse-reinforcement","slug":"risk-aware-active-inverse-reinforcement","title":"Risk-Aware Active Inverse Reinforcement Learning","date":"2019-01-08","arxiv_id":"1901.02161","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/risk-aware-active-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.02161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.02161"}},"official":{"repos":["Pearl-UTexas/ActiveVaR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/snas-stochastic-neural-architecture-search","slug":"snas-stochastic-neural-architecture-search","title":"SNAS: Stochastic Neural Architecture Search","date":"2018-12-24","arxiv_id":"1812.09926","repositories_listed":2,"syntology":null},{"url":"/paper/universal-successor-features-approximators","slug":"universal-successor-features-approximators","title":"Universal Successor Features Approximators","date":"2018-12-18","arxiv_id":"1812.07626","repositories_listed":2,"syntology":null},{"url":"/paper/decentralized-computation-offloading-for","slug":"decentralized-computation-offloading-for","title":"Decentralized Computation Offloading for Multi-User Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2018-12-16","arxiv_id":"1812.07394","repositories_listed":2,"syntology":null},{"url":"/paper/revisiting-the-softmax-bellman-operator","slug":"revisiting-the-softmax-bellman-operator","title":"Revisiting the Softmax Bellman Operator: New Benefits and New Perspective","date":"2018-12-02","arxiv_id":"1812.00456","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/revisiting-the-softmax-bellman-operator#ran","syntology_url":"https://syntology.ai/paper/1812.00456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.00456"}},"official":{"repos":["zhao-song/Softmax-DQN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/environments-for-lifelong-reinforcement","slug":"environments-for-lifelong-reinforcement","title":"Environments for Lifelong Reinforcement Learning","date":"2018-11-26","arxiv_id":"1811.10732","repositories_listed":2,"syntology":null},{"url":"/paper/learning-goal-embeddings-via-self-play-for","slug":"learning-goal-embeddings-via-self-play-for","title":"Learning Goal Embeddings via Self-Play for Hierarchical Reinforcement Learning","date":"2018-11-22","arxiv_id":"1811.09083","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-a-and-a-deep","slug":"reinforcement-learning-with-a-and-a-deep","title":"Reinforcement Learning with A* and a Deep Heuristic","date":"2018-11-19","arxiv_id":"1811.07745","repositories_listed":2,"syntology":null},{"url":"/paper/improving-automatic-source-code-summarization","slug":"improving-automatic-source-code-summarization","title":"Improving Automatic Source Code Summarization via Deep Reinforcement Learning","date":"2018-11-17","arxiv_id":"1811.07234","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-automatic-source-code-summarization#ran","syntology_url":"https://syntology.ai/paper/1811.07234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.07234"}},"official":null}},{"url":"/paper/reward-learning-from-human-preferences-and","slug":"reward-learning-from-human-preferences-and","title":"Reward learning from human preferences and demonstrations in Atari","date":"2018-11-15","arxiv_id":"1811.06521","repositories_listed":2,"syntology":null},{"url":"/paper/natural-environment-benchmarks-for","slug":"natural-environment-benchmarks-for","title":"Natural Environment Benchmarks for Reinforcement Learning","date":"2018-11-14","arxiv_id":"1811.06032","repositories_listed":2,"syntology":null},{"url":"/paper/a-hierarchical-framework-for-relation","slug":"a-hierarchical-framework-for-relation","title":"A Hierarchical Framework for Relation Extraction with Reinforcement Learning","date":"2018-11-09","arxiv_id":"1811.03925","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-automatic-test","slug":"reinforcement-learning-for-automatic-test","title":"Reinforcement Learning for Automatic Test Case Prioritization and Selection in Continuous Integration","date":"2018-11-09","arxiv_id":"1811.04122","repositories_listed":2,"syntology":null},{"url":"/paper/memory-based-deep-reinforcement-learning-for","slug":"memory-based-deep-reinforcement-learning-for","title":"Memory-based Deep Reinforcement Learning for Obstacle Avoidance in UAV with Limited Environment Knowledge","date":"2018-11-08","arxiv_id":"1811.03307","repositories_listed":2,"syntology":null},{"url":"/paper/horizon-facebooks-open-source-applied","slug":"horizon-facebooks-open-source-applied","title":"Horizon: Facebook's Open Source Applied Reinforcement Learning Platform","date":"2018-11-01","arxiv_id":"1811.00260","repositories_listed":2,"syntology":null},{"url":"/paper/temporal-regularization-in-markov-decision","slug":"temporal-regularization-in-markov-decision","title":"Temporal Regularization in Markov Decision Process","date":"2018-11-01","arxiv_id":"1811.00429","repositories_listed":2,"syntology":null},{"url":"/paper/model-based-active-exploration","slug":"model-based-active-exploration","title":"Model-Based Active Exploration","date":"2018-10-29","arxiv_id":"1810.12162","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-active-exploration#ran","syntology_url":"https://syntology.ai/paper/1810.12162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12162"}},"official":{"repos":["nnaisense/max"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-modular-control-for-embodied-question","slug":"neural-modular-control-for-embodied-question","title":"Neural Modular Control for Embodied Question Answering","date":"2018-10-26","arxiv_id":"1810.11181","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/neural-modular-control-for-embodied-question#ran","syntology_url":"https://syntology.ai/paper/1810.11181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11181"}},"official":null}},{"url":"/paper/making-sense-of-vision-and-touch-self","slug":"making-sense-of-vision-and-touch-self","title":"Making Sense of Vision and Touch: Self-Supervised Learning of Multimodal Representations for Contact-Rich Tasks","date":"2018-10-24","arxiv_id":"1810.10191","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/making-sense-of-vision-and-touch-self#ran","syntology_url":"https://syntology.ai/paper/1810.10191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.10191"}},"official":{"repos":["stanford-iprl-lab/multimodal_representation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-deep-reinforcement-learning-using-online","slug":"fast-deep-reinforcement-learning-using-online","title":"Fast deep reinforcement learning using online adjustments from the past","date":"2018-10-18","arxiv_id":"1810.08163","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-deep-reinforcement-learning-using-online#ran","syntology_url":"https://syntology.ai/paper/1810.08163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.08163"}},"official":null}},{"url":"/paper/successor-uncertainties-exploration-and","slug":"successor-uncertainties-exploration-and","title":"Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","date":"2018-10-15","arxiv_id":"1810.06530","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/successor-uncertainties-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1810.06530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06530"}},"official":null}},{"url":"/paper/q-map-a-convolutional-approach-for-goal","slug":"q-map-a-convolutional-approach-for-goal","title":"Scaling All-Goals Updates in Reinforcement Learning Using Convolutional Neural Networks","date":"2018-10-06","arxiv_id":"1810.02927","repositories_listed":2,"syntology":null},{"url":"/paper/learning-scheduling-algorithms-for-data","slug":"learning-scheduling-algorithms-for-data","title":"Learning Scheduling Algorithms for Data Processing Clusters","date":"2018-10-03","arxiv_id":"1810.01963","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-scheduling-algorithms-for-data#ran","syntology_url":"https://syntology.ai/paper/1810.01963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01963"}},"official":{"repos":["hongzimao/decima-sim"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/efficient-dialog-policy-learning-via-positive","slug":"efficient-dialog-policy-learning-via-positive","title":"Efficient Dialog Policy Learning via Positive Memory Retention","date":"2018-10-02","arxiv_id":"1810.01371","repositories_listed":2,"syntology":null},{"url":"/paper/energy-based-hindsight-experience","slug":"energy-based-hindsight-experience","title":"Energy-Based Hindsight Experience Prioritization","date":"2018-10-02","arxiv_id":"1810.01363","repositories_listed":2,"syntology":null},{"url":"/paper/solving-statistical-mechanics-using","slug":"solving-statistical-mechanics-using","title":"Solving Statistical Mechanics Using Variational Autoregressive Networks","date":"2018-09-27","arxiv_id":"1809.10606","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/solving-statistical-mechanics-using#ran","syntology_url":"https://syntology.ai/paper/1809.10606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.10606"}},"official":{"repos":["wangleiphy/VAN.jl","wdphy16/stat-mech-van"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-reinforcement-learning","slug":"benchmarking-reinforcement-learning","title":"Benchmarking Reinforcement Learning Algorithms on Real-World Robots","date":"2018-09-20","arxiv_id":"1809.07731","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1809.07731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.07731"}},"official":{"repos":["kindredresearch/SenseAct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with","slug":"multi-task-deep-reinforcement-learning-with","title":"Multi-task Deep Reinforcement Learning with PopArt","date":"2018-09-12","arxiv_id":"1809.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1809.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.04474"}},"official":null}},{"url":"/paper/apes-a-python-toolbox-for-simulating","slug":"apes-a-python-toolbox-for-simulating","title":"APES: a Python toolbox for simulating reinforcement learning environments","date":"2018-08-31","arxiv_id":"1808.10692","repositories_listed":2,"syntology":null},{"url":"/paper/application-of-self-play-reinforcement","slug":"application-of-self-play-reinforcement","title":"Application of Self-Play Reinforcement Learning to a Four-Player Game of Imperfect Information","date":"2018-08-30","arxiv_id":"1808.10442","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-for-relation","slug":"reinforcement-learning-for-relation","title":"Reinforcement Learning for Relation Classification from Noisy Data","date":"2018-08-24","arxiv_id":"1808.08013","repositories_listed":2,"syntology":null},{"url":"/paper/a-framework-for-automated-cellular-network","slug":"a-framework-for-automated-cellular-network","title":"A Framework for Automated Cellular Network Tuning with Reinforcement Learning","date":"2018-08-13","arxiv_id":"1808.05140","repositories_listed":2,"syntology":null},{"url":"/paper/count-based-exploration-with-the-successor","slug":"count-based-exploration-with-the-successor","title":"Count-Based Exploration with the Successor Representation","date":"2018-07-31","arxiv_id":"1807.11622","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/count-based-exploration-with-the-successor#ran","syntology_url":"https://syntology.ai/paper/1807.11622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.11622"}},"official":{"repos":["mcmachado/count_based_exploration_sr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/remember-and-forget-for-experience-replay","slug":"remember-and-forget-for-experience-replay","title":"Remember and Forget for Experience Replay","date":"2018-07-16","arxiv_id":"1807.05827","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/remember-and-forget-for-experience-replay#ran","syntology_url":"https://syntology.ai/paper/1807.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.05827"}},"official":{"repos":["cselab/smarties"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-reinforcement-learning-with-imagined","slug":"visual-reinforcement-learning-with-imagined","title":"Visual Reinforcement Learning with Imagined Goals","date":"2018-07-12","arxiv_id":"1807.04742","repositories_listed":2,"syntology":null},{"url":"/paper/algorithmic-framework-for-model-based-deep","slug":"algorithmic-framework-for-model-based-deep","title":"Algorithmic Framework for Model-based Deep Reinforcement Learning with Theoretical Guarantees","date":"2018-07-10","arxiv_id":"1807.03858","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/algorithmic-framework-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1807.03858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03858"}},"official":{"repos":["roosephu/slbo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ranked-reward-enabling-self-play","slug":"ranked-reward-enabling-self-play","title":"Ranked Reward: Enabling Self-Play Reinforcement Learning for Combinatorial Optimization","date":"2018-07-04","arxiv_id":"1807.01672","repositories_listed":2,"syntology":null},{"url":"/paper/sample-efficient-reinforcement-learning-with","slug":"sample-efficient-reinforcement-learning-with","title":"Sample-Efficient Reinforcement Learning with Stochastic Ensemble Value Expansion","date":"2018-07-04","arxiv_id":"1807.01675","repositories_listed":2,"syntology":null},{"url":"/paper/accuracy-based-curriculum-learning-in-deep","slug":"accuracy-based-curriculum-learning-in-deep","title":"Accuracy-based Curriculum Learning in Deep Reinforcement Learning","date":"2018-06-25","arxiv_id":"1806.09614","repositories_listed":2,"syntology":null},{"url":"/paper/rudder-return-decomposition-for-delayed","slug":"rudder-return-decomposition-for-delayed","title":"RUDDER: Return Decomposition for Delayed Rewards","date":"2018-06-20","arxiv_id":"1806.07857","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rudder-return-decomposition-for-delayed#ran","syntology_url":"https://syntology.ai/paper/1806.07857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07857"}},"official":{"repos":["ml-jku/baselines-rudder","ml-jku/rudder"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-variational-learning-of-bayesian","slug":"structured-variational-learning-of-bayesian","title":"Structured Variational Learning of Bayesian Neural Networks with Horseshoe Priors","date":"2018-06-13","arxiv_id":"1806.05975","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-variational-learning-of-bayesian#ran","syntology_url":"https://syntology.ai/paper/1806.05975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.05975"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/bayesian-inference-with-anchored-ensembles-of","slug":"bayesian-inference-with-anchored-ensembles-of","title":"Bayesian Inference with Anchored Ensembles of Neural Networks, and Application to Exploration in Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11324","repositories_listed":2,"syntology":null},{"url":"/paper/virtual-taobao-virtualizing-real-world-online","slug":"virtual-taobao-virtualizing-real-world-online","title":"Virtual-Taobao: Virtualizing Real-world Online Retail Environment for Reinforcement Learning","date":"2018-05-25","arxiv_id":"1805.10000","repositories_listed":2,"syntology":null},{"url":"/paper/visceral-machines-reinforcement-learning-with","slug":"visceral-machines-reinforcement-learning-with","title":"Visceral Machines: Risk-Aversion in Reinforcement Learning with Intrinsic Physiological Rewards","date":"2018-05-25","arxiv_id":"1805.09975","repositories_listed":2,"syntology":null},{"url":"/paper/a0c-alpha-zero-in-continuous-action-space","slug":"a0c-alpha-zero-in-continuous-action-space","title":"A0C: Alpha Zero in Continuous Action Space","date":"2018-05-24","arxiv_id":"1805.09613","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a0c-alpha-zero-in-continuous-action-space#ran","syntology_url":"https://syntology.ai/paper/1805.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09613"}},"official":null}},{"url":"/paper/robust-distant-supervision-relation","slug":"robust-distant-supervision-relation","title":"Robust Distant Supervision Relation Extraction via Deep Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09927","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-distant-supervision-relation#ran","syntology_url":"https://syntology.ai/paper/1805.09927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09927"}},"official":null}},{"url":"/paper/verifiable-reinforcement-learning-via-policy","slug":"verifiable-reinforcement-learning-via-policy","title":"Verifiable Reinforcement Learning via Policy Extraction","date":"2018-05-22","arxiv_id":"1805.08328","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-and-control-as","slug":"reinforcement-learning-and-control-as","title":"Reinforcement Learning and Control as Probabilistic Inference: Tutorial and Review","date":"2018-05-02","arxiv_id":"1805.00909","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-and-control-as#ran","syntology_url":"https://syntology.ai/paper/1805.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.00909"}},"official":null}},{"url":"/paper/dora-the-explorer-directed-outreaching","slug":"dora-the-explorer-directed-outreaching","title":"DORA The Explorer: Directed Outreaching Reinforcement Action-Selection","date":"2018-04-11","arxiv_id":"1804.04012","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dora-the-explorer-directed-outreaching#ran","syntology_url":"https://syntology.ai/paper/1804.04012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.04012"}},"official":{"repos":["borgr/DORA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/crafting-a-toolchain-for-image-restoration-by","slug":"crafting-a-toolchain-for-image-restoration-by","title":"Crafting a Toolchain for Image Restoration by Deep Reinforcement Learning","date":"2018-04-10","arxiv_id":"1804.03312","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-run-challenge-solutions-adapting","slug":"learning-to-run-challenge-solutions-adapting","title":"Learning to Run challenge solutions: Adapting reinforcement learning methods for neuromusculoskeletal environments","date":"2018-04-02","arxiv_id":"1804.00361","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-run-challenge-solutions-adapting#ran","syntology_url":"https://syntology.ai/paper/1804.00361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.00361"}},"official":{"repos":["AdamStelmaszczyk/learning2run"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-adapt-in-dynamic-real-world","slug":"learning-to-adapt-in-dynamic-real-world","title":"Learning to Adapt in Dynamic, Real-World Environments Through Meta-Reinforcement Learning","date":"2018-03-30","arxiv_id":"1803.11347","repositories_listed":2,"syntology":null}],"record_sha256":"480fda1c1116f470be0003ad82d79bbd12bc04f72ffa03435ae6e19e79e07ce3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}