{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/37","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":37,"pages_in_order":132,"rows_per_page":100,"rows":[3601,3700],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/36","next":"/task/reinforcement-learning/papers/38","papers":[{"url":"/paper/reinforcement-learning-with-dynamic-boltzmann","slug":"reinforcement-learning-with-dynamic-boltzmann","title":"Reinforcement Learning with Dynamic Boltzmann Softmax Updates","date":"2019-03-14","arxiv_id":"1903.05926","repositories_listed":1,"syntology":null},{"url":"/paper/coacor-code-annotation-for-code-retrieval","slug":"coacor-code-annotation-for-code-retrieval","title":"CoaCor: Code Annotation for Code Retrieval with Reinforcement Learning","date":"2019-03-13","arxiv_id":"1904.00720","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/coacor-code-annotation-for-code-retrieval#ran","syntology_url":"https://syntology.ai/paper/1904.00720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.00720"}},"official":{"repos":["LittleYUYU/CoaCor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/vrkitchen-an-interactive-3d-virtual","slug":"vrkitchen-an-interactive-3d-virtual","title":"VRKitchen: an Interactive 3D Virtual Environment for Task-oriented Learning","date":"2019-03-13","arxiv_id":"1903.05757","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-pitfalls-of-measuring-emergent","slug":"on-the-pitfalls-of-measuring-emergent","title":"On the Pitfalls of Measuring Emergent Communication","date":"2019-03-12","arxiv_id":"1903.05168","repositories_listed":1,"syntology":null},{"url":"/paper/universally-slimmable-networks-and-improved","slug":"universally-slimmable-networks-and-improved","title":"Universally Slimmable Networks and Improved Training Techniques","date":"2019-03-12","arxiv_id":"1903.05134","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/universally-slimmable-networks-and-improved#ran","syntology_url":"https://syntology.ai/paper/1903.05134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.05134"}},"official":{"repos":["JiahuiYu/slimmable_networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-reinforcement-learning-with-expert","slug":"hybrid-reinforcement-learning-with-expert","title":"Hybrid Reinforcement Learning with Expert State Sequences","date":"2019-03-11","arxiv_id":"1903.04110","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-reinforcement-learning-with-expert#ran","syntology_url":"https://syntology.ai/paper/1903.04110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04110"}},"official":{"repos":["XiaoxiaoGuo/tensor4rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-2","slug":"multi-agent-deep-reinforcement-learning-for-2","title":"Multi-Agent Deep Reinforcement Learning for Large-scale Traffic Signal Control","date":"2019-03-11","arxiv_id":"1903.04527","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-deep-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/1903.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04527"}},"official":{"repos":["cts198859/deeprl_signal_control"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-model-free-reinforcement","slug":"sample-efficient-model-free-reinforcement","title":"Sample-Efficient Model-Free Reinforcement Learning with Off-Policy Critics","date":"2019-03-11","arxiv_id":"1903.04193","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-power-system-emergency-control-using","slug":"adaptive-power-system-emergency-control-using","title":"Adaptive Power System Emergency Control using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03712","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-power-system-emergency-control-using#ran","syntology_url":"https://syntology.ai/paper/1903.03712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03712"}},"official":{"repos":["RLGC-Project/RLGC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-sample-efficient-blackbox","slug":"adaptive-sample-efficient-blackbox","title":"From Complexity to Simplicity: Adaptive ES-Active Subspaces for Blackbox Optimization","date":"2019-03-07","arxiv_id":"1903.04268","repositories_listed":1,"syntology":null},{"url":"/paper/concurrent-meta-reinforcement-learning","slug":"concurrent-meta-reinforcement-learning","title":"Concurrent Meta Reinforcement Learning","date":"2019-03-07","arxiv_id":"1903.02710","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-for-channel-coding-via-neural","slug":"deep-learning-for-channel-coding-via-neural","title":"Deep Learning for Channel Coding via Neural Mutual Information Estimation","date":"2019-03-07","arxiv_id":"1903.02865","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-research-trends-from-arxiv","slug":"predicting-research-trends-from-arxiv","title":"Predicting Research Trends From Arxiv","date":"2019-03-07","arxiv_id":"1903.02831","repositories_listed":1,"syntology":null},{"url":"/paper/a-hitchhiker-s-guide-to-statistical","slug":"a-hitchhiker-s-guide-to-statistical","title":"A Hitchhiker's Guide to Statistical Comparisons of Reinforcement Learning Algorithms","date":"2019-03-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/simple-rl-reproducible-reinforcement-learning","slug":"simple-rl-reproducible-reinforcement-learning","title":"simple_rl: Reproducible Reinforcement Learning in Python","date":"2019-03-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-active-localization","slug":"deep-active-localization","title":"Deep Active Localization","date":"2019-03-05","arxiv_id":"1903.01669","repositories_listed":1,"syntology":null},{"url":"/paper/using-natural-language-for-reward-shaping-in","slug":"using-natural-language-for-reward-shaping-in","title":"Using Natural Language for Reward Shaping in Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.02020","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/using-natural-language-for-reward-shaping-in#ran","syntology_url":"https://syntology.ai/paper/1903.02020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.02020"}},"official":null}},{"url":"/paper/viewpoint-optimization-for-autonomous","slug":"viewpoint-optimization-for-autonomous","title":"Viewpoint Optimization for Autonomous Strawberry Harvesting with Deep Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.02074","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-actor-critic-reinforcement-learning-in","slug":"hybrid-actor-critic-reinforcement-learning-in","title":"Hybrid Actor-Critic Reinforcement Learning in Parameterized Action Space","date":"2019-03-04","arxiv_id":"1903.01344","repositories_listed":1,"syntology":null},{"url":"/paper/model-primitive-hierarchical-lifelong","slug":"model-primitive-hierarchical-lifelong","title":"Model Primitive Hierarchical Lifelong Reinforcement Learning","date":"2019-03-04","arxiv_id":"1903.01567","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-primitive-hierarchical-lifelong#ran","syntology_url":"https://syntology.ai/paper/1903.01567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.01567"}},"official":{"repos":["sisl/MPHRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/norml-no-reward-meta-learning","slug":"norml-no-reward-meta-learning","title":"NoRML: No-Reward Meta Learning","date":"2019-03-04","arxiv_id":"1903.01063","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-episodic-deep-deterministic","slug":"asynchronous-episodic-deep-deterministic","title":"Asynchronous Episodic Deep Deterministic Policy Gradient: Towards Continuous Control in Computationally Complex Environments","date":"2019-03-03","arxiv_id":"1903.00827","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-up-budgeted-reinforcement-learning","slug":"scaling-up-budgeted-reinforcement-learning","title":"Budgeted Reinforcement Learning in Continuous State Space","date":"2019-03-03","arxiv_id":"1903.01004","repositories_listed":1,"syntology":null},{"url":"/paper/a-cooperative-multi-agent-reinforcement","slug":"a-cooperative-multi-agent-reinforcement","title":"A Cooperative Multi-Agent Reinforcement Learning Framework for Resource Balancing in Complex Logistics Network","date":"2019-03-02","arxiv_id":"1903.00714","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reinforcement-learning-with-a-mind","slug":"efficient-reinforcement-learning-with-a-mind","title":"Efficient Reinforcement Learning for StarCraft by Abstract Forward Models and Transfer Learning","date":"2019-03-02","arxiv_id":"1903.00715","repositories_listed":1,"syntology":null},{"url":"/paper/catalystrl-a-distributed-framework-for","slug":"catalystrl-a-distributed-framework-for","title":"Catalyst.RL: A Distributed Framework for Reproducible RL Research","date":"2019-02-28","arxiv_id":"1903.00027","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-rewards-for-question-generation","slug":"evaluating-rewards-for-question-generation","title":"Evaluating Rewards for Question Generation Models","date":"2019-02-28","arxiv_id":"1902.11049","repositories_listed":1,"syntology":null},{"url":"/paper/regularity-normalization-constraining","slug":"regularity-normalization-constraining","title":"Unsupervised Attention Mechanism across Neural Network Layers","date":"2019-02-27","arxiv_id":"1902.10658","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-bottlenecks-in-deep-q-learning","slug":"diagnosing-bottlenecks-in-deep-q-learning","title":"Diagnosing Bottlenecks in Deep Q-learning Algorithms","date":"2019-02-26","arxiv_id":"1902.10250","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diagnosing-bottlenecks-in-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1902.10250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.10250"}},"official":null}},{"url":"/paper/function-space-particle-optimization-for","slug":"function-space-particle-optimization-for","title":"Function Space Particle Optimization for Bayesian Neural Networks","date":"2019-02-26","arxiv_id":"1902.09754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/function-space-particle-optimization-for#ran","syntology_url":"https://syntology.ai/paper/1902.09754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.09754"}},"official":{"repos":["thu-ml/fpovi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flappy-hummingbird-an-open-source-dynamic","slug":"flappy-hummingbird-an-open-source-dynamic","title":"Flappy Hummingbird: An Open Source Dynamic Simulation of Flapping Wing Robots and Animals","date":"2019-02-25","arxiv_id":"1902.09628","repositories_listed":1,"syntology":null},{"url":"/paper/marathon-environments-multi-agent-continuous","slug":"marathon-environments-multi-agent-continuous","title":"Marathon Environments: Multi-Agent Continuous Control Benchmarks in a Modern Video Game Engine","date":"2019-02-25","arxiv_id":"1902.09097","repositories_listed":1,"syntology":null},{"url":"/paper/a-general-framework-for-structured-learning","slug":"a-general-framework-for-structured-learning","title":"A General Framework for Structured Learning of Mechanical Systems","date":"2019-02-22","arxiv_id":"1902.08705","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-general-framework-for-structured-learning#ran","syntology_url":"https://syntology.ai/paper/1902.08705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.08705"}},"official":{"repos":["sisl/mechamodlearn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-grounding-of-plannable-first","slug":"unsupervised-grounding-of-plannable-first","title":"Unsupervised Grounding of Plannable First-Order Logic Representation from Images","date":"2019-02-21","arxiv_id":"1902.08093","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-confidence-regions-tight-bayesian","slug":"beyond-confidence-regions-tight-bayesian","title":"Beyond Confidence Regions: Tight Bayesian Ambiguity Sets for Robust MDPs","date":"2019-02-20","arxiv_id":"1902.07605","repositories_listed":1,"syntology":null},{"url":"/paper/dom-q-net-grounded-rl-on-structured-language","slug":"dom-q-net-grounded-rl-on-structured-language","title":"DOM-Q-NET: Grounded RL on Structured Language","date":"2019-02-19","arxiv_id":"1902.07257","repositories_listed":1,"syntology":null},{"url":"/paper/hyperbolic-discounting-and-learning-over","slug":"hyperbolic-discounting-and-learning-over","title":"Hyperbolic Discounting and Learning over Multiple Horizons","date":"2019-02-19","arxiv_id":"1902.06865","repositories_listed":1,"syntology":null},{"url":"/paper/heuristics-answer-set-programming-and-markov","slug":"heuristics-answer-set-programming-and-markov","title":"Heuristics, Answer Set Programming and Markov Decision Process for Solving a Set of Spatial Puzzles","date":"2019-02-16","arxiv_id":"1903.03411","repositories_listed":1,"syntology":null},{"url":"/paper/prolonets-neural-encoding-human-experts","slug":"prolonets-neural-encoding-human-experts","title":"Neural-encoding Human Experts' Domain Knowledge to Warm Start Reinforcement Learning","date":"2019-02-15","arxiv_id":"1902.06007","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-visuomotor-control-through","slug":"unsupervised-visuomotor-control-through","title":"Unsupervised Visuomotor Control through Distributional Planning Networks","date":"2019-02-14","arxiv_id":"1902.05542","repositories_listed":1,"syntology":null},{"url":"/paper/verifiably-safe-off-model-reinforcement","slug":"verifiably-safe-off-model-reinforcement","title":"Verifiably Safe Off-Model Reinforcement Learning","date":"2019-02-14","arxiv_id":"1902.05632","repositories_listed":1,"syntology":null},{"url":"/paper/preferences-implicit-in-the-state-of-the","slug":"preferences-implicit-in-the-state-of-the","title":"Preferences Implicit in the State of the World","date":"2019-02-12","arxiv_id":"1902.04198","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preferences-implicit-in-the-state-of-the#ran","syntology_url":"https://syntology.ai/paper/1902.04198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.04198"}},"official":{"repos":["HumanCompatibleAI/rlsp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-through-simulation-integrating","slug":"generalization-through-simulation-integrating","title":"Generalization through Simulation: Integrating Simulated and Real Data into Deep Reinforcement Learning for Vision-Based Autonomous Flight","date":"2019-02-11","arxiv_id":"1902.03701","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generalization-through-simulation-integrating#ran","syntology_url":"https://syntology.ai/paper/1902.03701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.03701"}},"official":{"repos":["gkahn13/GtS"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/novelty-search-for-deep-reinforcement","slug":"novelty-search-for-deep-reinforcement","title":"Novelty Search for Deep Reinforcement Learning Policy Network Weights by Action Sequence Edit Metric Distance","date":"2019-02-08","arxiv_id":"1902.03142","repositories_listed":1,"syntology":null},{"url":"/paper/artificial-intelligence-for-prosthetics","slug":"artificial-intelligence-for-prosthetics","title":"Artificial Intelligence for Prosthetics - challenge solutions","date":"2019-02-07","arxiv_id":"1902.02441","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/artificial-intelligence-for-prosthetics#ran","syntology_url":"https://syntology.ai/paper/1902.02441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.02441"}},"official":{"repos":["iasawseen/MultiServerRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deeper-sparser-exploration","slug":"deeper-sparser-exploration","title":"Bayesian Reinforcement Learning via Deep, Sparse Sampling","date":"2019-02-07","arxiv_id":"1902.02661","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-learning-components-for-safety-in","slug":"augmenting-learning-components-for-safety-in","title":"Dynamic-Weighted Simplex Strategy for Learning Enabled Cyber Physical Systems","date":"2019-02-06","arxiv_id":"1902.02432","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-schedule-communication-in-multi","slug":"learning-to-schedule-communication-in-multi","title":"Learning to Schedule Communication in Multi-agent Reinforcement Learning","date":"2019-02-05","arxiv_id":"1902.01554","repositories_listed":1,"syntology":null},{"url":"/paper/separating-value-functions-across-time-scales","slug":"separating-value-functions-across-time-scales","title":"Separating value functions across time-scales","date":"2019-02-05","arxiv_id":"1902.01883","repositories_listed":1,"syntology":null},{"url":"/paper/the-natural-language-of-actions","slug":"the-natural-language-of-actions","title":"The Natural Language of Actions","date":"2019-02-04","arxiv_id":"1902.01119","repositories_listed":1,"syntology":null},{"url":"/paper/a-meta-mdp-approach-to-exploration-for","slug":"a-meta-mdp-approach-to-exploration-for","title":"A Meta-MDP Approach to Exploration for Lifelong Reinforcement Learning","date":"2019-02-03","arxiv_id":"1902.00843","repositories_listed":1,"syntology":null},{"url":"/paper/certified-reinforcement-learning-with-logic","slug":"certified-reinforcement-learning-with-logic","title":"Certified Reinforcement Learning with Logic Guidance","date":"2019-02-02","arxiv_id":"1902.00778","repositories_listed":1,"syntology":null},{"url":"/paper/minmax-optimization-stable-limit-points-of","slug":"minmax-optimization-stable-limit-points-of","title":"What is Local Optimality in Nonconvex-Nonconcave Minimax Optimization?","date":"2019-02-02","arxiv_id":"1902.00618","repositories_listed":1,"syntology":null},{"url":"/paper/policy-consolidation-for-continual","slug":"policy-consolidation-for-continual","title":"Policy Consolidation for Continual Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00255","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-consolidation-for-continual#ran","syntology_url":"https://syntology.ai/paper/1902.00255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00255"}},"official":null}},{"url":"/paper/sequential-evaluation-and-generation","slug":"sequential-evaluation-and-generation","title":"Sequential Evaluation and Generation Framework for Combinatorial Recommender System","date":"2019-02-01","arxiv_id":"1902.00245","repositories_listed":1,"syntology":null},{"url":"/paper/tf-replicator-distributed-machine-learning","slug":"tf-replicator-distributed-machine-learning","title":"TF-Replicator: Distributed Machine Learning for Researchers","date":"2019-02-01","arxiv_id":"1902.00465","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tf-replicator-distributed-machine-learning#ran","syntology_url":"https://syntology.ai/paper/1902.00465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00465"}},"official":{"repos":["tensorflow/community"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrasting-exploration-in-parameter-and","slug":"contrasting-exploration-in-parameter-and","title":"Contrasting Exploration in Parameter and Action Space: A Zeroth-Order Optimization Perspective","date":"2019-01-31","arxiv_id":"1901.11503","repositories_listed":1,"syntology":null},{"url":"/paper/private-q-learning-with-functional-noise-in","slug":"private-q-learning-with-functional-noise-in","title":"Privacy-preserving Q-Learning with Functional Noise in Continuous State Spaces","date":"2019-01-30","arxiv_id":"1901.10634","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-hierarchy-via-reinforcement","slug":"emergence-of-hierarchy-via-reinforcement","title":"Self-organization of action hierarchy and compositionality by reinforcement learning with recurrent neural networks","date":"2019-01-29","arxiv_id":"1901.10113","repositories_listed":1,"syntology":null},{"url":"/paper/safe-efficient-and-comfortable-velocity","slug":"safe-efficient-and-comfortable-velocity","title":"Safe, Efficient, and Comfortable Velocity Control based on Reinforcement Learning for Autonomous Driving","date":"2019-01-29","arxiv_id":"1902.00089","repositories_listed":1,"syntology":null},{"url":"/paper/lyapunov-based-safe-policy-optimization-for","slug":"lyapunov-based-safe-policy-optimization-for","title":"Lyapunov-based Safe Policy Optimization for Continuous Control","date":"2019-01-28","arxiv_id":"1901.10031","repositories_listed":1,"syntology":null},{"url":"/paper/making-deep-q-learning-methods-robust-to-time","slug":"making-deep-q-learning-methods-robust-to-time","title":"Making Deep Q-learning methods robust to time discretization","date":"2019-01-28","arxiv_id":"1901.09732","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-deep-q-learning-methods-robust-to-time#ran","syntology_url":"https://syntology.ai/paper/1901.09732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09732"}},"official":null}},{"url":"/paper/multi-agent-generalized-recursive-reasoning","slug":"multi-agent-generalized-recursive-reasoning","title":"Modelling Bounded Rationality in Multi-Agent Interactions by Generalized Recursive Reasoning","date":"2019-01-26","arxiv_id":"1901.09216","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multi-agent-generalized-recursive-reasoning#ran","syntology_url":"https://syntology.ai/paper/1901.09216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09216"}},"official":{"repos":["ying-wen/gr2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-linguistic-phenomena-in-multi-agent","slug":"emergent-linguistic-phenomena-in-multi-agent","title":"Emergent Linguistic Phenomena in Multi-Agent Communication Games","date":"2019-01-25","arxiv_id":"1901.08706","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/emergent-linguistic-phenomena-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1901.08706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08706"}},"official":{"repos":["lgraesser/MultimodalGame"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/combinational-q-learning-for-dou-di-zhu","slug":"combinational-q-learning-for-dou-di-zhu","title":"Combinational Q-Learning for Dou Di Zhu","date":"2019-01-24","arxiv_id":"1901.08925","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-measurement-scheduling-for-event","slug":"dynamic-measurement-scheduling-for-event","title":"Dynamic Measurement Scheduling for Event Forecasting using Deep RL","date":"2019-01-24","arxiv_id":"1901.09699","repositories_listed":1,"syntology":null},{"url":"/paper/fairness-with-dynamics","slug":"fairness-with-dynamics","title":"Algorithms for Fairness in Sequential Decision Making","date":"2019-01-24","arxiv_id":"1901.08568","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fairness-with-dynamics#ran","syntology_url":"https://syntology.ai/paper/1901.08568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08568"}},"official":{"repos":["wmgithub/fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-reasoning-from-meta-reinforcement","slug":"causal-reasoning-from-meta-reinforcement","title":"Causal Reasoning from Meta-reinforcement Learning","date":"2019-01-23","arxiv_id":"1901.08162","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causal-reasoning-from-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.08162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08162"}},"official":null}},{"url":"/paper/understanding-multi-step-deep-reinforcement","slug":"understanding-multi-step-deep-reinforcement","title":"Understanding Multi-Step Deep Reinforcement Learning: A Systematic Study of the DQN Target","date":"2019-01-22","arxiv_id":"1901.07510","repositories_listed":1,"syntology":null},{"url":"/paper/read-watch-and-move-reinforcement-learning","slug":"read-watch-and-move-reinforcement-learning","title":"Read, Watch, and Move: Reinforcement Learning for Temporally Grounding Natural Language Descriptions in Videos","date":"2019-01-21","arxiv_id":"1901.06829","repositories_listed":1,"syntology":null},{"url":"/paper/wall-e-an-efficient-reinforcement-learning","slug":"wall-e-an-efficient-reinforcement-learning","title":"WALL-E: An Efficient Reinforcement Learning Research Framework","date":"2019-01-18","arxiv_id":"1901.06086","repositories_listed":1,"syntology":null},{"url":"/paper/amplifying-the-imitation-effect-for","slug":"amplifying-the-imitation-effect-for","title":"Amplifying the Imitation Effect for Reinforcement Learning of UCAV's Mission Execution","date":"2019-01-17","arxiv_id":"1901.05856","repositories_listed":1,"syntology":null},{"url":"/paper/autophase-compiler-phase-ordering-for-high","slug":"autophase-compiler-phase-ordering-for-high","title":"AutoPhase: Compiler Phase-Ordering for High Level Synthesis with Deep Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04615","repositories_listed":1,"syntology":null},{"url":"/paper/transfer-learning-for-prosthetics-using","slug":"transfer-learning-for-prosthetics-using","title":"Transfer Learning for Prosthetics Using Imitation Learning","date":"2019-01-15","arxiv_id":"1901.04772","repositories_listed":1,"syntology":null},{"url":"/paper/improving-coordination-in-multi-agent-deep","slug":"improving-coordination-in-multi-agent-deep","title":"Improving Coordination in Small-Scale Multi-Agent Deep Reinforcement Learning through Memory-driven Communication","date":"2019-01-12","arxiv_id":"1901.03887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-coordination-in-multi-agent-deep#ran","syntology_url":"https://syntology.ai/paper/1901.03887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.03887"}},"official":null}},{"url":"/paper/an-investigation-of-model-free-planning","slug":"an-investigation-of-model-free-planning","title":"An investigation of model-free planning","date":"2019-01-11","arxiv_id":"1901.03559","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-via","slug":"hierarchical-reinforcement-learning-via","title":"Hierarchical Reinforcement Learning via Advantage-Weighted Information Maximization","date":"2019-01-05","arxiv_id":"1901.01365","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-learning-of-image-embedding","slug":"self-supervised-learning-of-image-embedding","title":"Self-supervised Learning of Image Embedding for Continuous Control","date":"2019-01-03","arxiv_id":"1901.00943","repositories_listed":1,"syntology":null},{"url":"/paper/opportunistic-learning-budgeted-cost","slug":"opportunistic-learning-budgeted-cost","title":"Opportunistic Learning: Budgeted Cost-Sensitive Learning from Data Streams","date":"2019-01-02","arxiv_id":"1901.00243","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opportunistic-learning-budgeted-cost#ran","syntology_url":"https://syntology.ai/paper/1901.00243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.00243"}},"official":{"repos":["mkachuee/Opportunistic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mid-level-visual-representations-improve","slug":"mid-level-visual-representations-improve","title":"Mid-Level Visual Representations Improve Generalization and Sample Efficiency for Learning Visuomotor Policies","date":"2018-12-31","arxiv_id":"1812.11971","repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-interpret-atari-agents","slug":"learn-to-interpret-atari-agents","title":"Learn to Interpret Atari Agents","date":"2018-12-29","arxiv_id":"1812.11276","repositories_listed":1,"syntology":null},{"url":"/paper/generative-adversarial-user-model-for","slug":"generative-adversarial-user-model-for","title":"Generative Adversarial User Model for Reinforcement Learning Based Recommendation System","date":"2018-12-27","arxiv_id":"1812.10613","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generative-adversarial-user-model-for#ran","syntology_url":"https://syntology.ai/paper/1812.10613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.10613"}},"official":{"repos":["xinshi-chen/GenerativeAdversarialUserModel"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deconfounding-reinforcement-learning-in","slug":"deconfounding-reinforcement-learning-in","title":"Deconfounding Reinforcement Learning in Observational Settings","date":"2018-12-26","arxiv_id":"1812.10576","repositories_listed":1,"syntology":null},{"url":"/paper/iroko-a-framework-to-prototype-reinforcement","slug":"iroko-a-framework-to-prototype-reinforcement","title":"Iroko: A Framework to Prototype Reinforcement Learning for Data Center Traffic Control","date":"2018-12-24","arxiv_id":"1812.09975","repositories_listed":1,"syntology":null},{"url":"/paper/chamnet-towards-efficient-network-design","slug":"chamnet-towards-efficient-network-design","title":"ChamNet: Towards Efficient Network Design through Platform-Aware Model Adaptation","date":"2018-12-21","arxiv_id":"1812.08934","repositories_listed":1,"syntology":null},{"url":"/paper/introducing-neuromodulation-in-deep-neural","slug":"introducing-neuromodulation-in-deep-neural","title":"Introducing Neuromodulation in Deep Neural Networks to Learn Adaptive Behaviours","date":"2018-12-21","arxiv_id":"1812.09113","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-with-non-expert-human","slug":"pre-training-with-non-expert-human","title":"Pre-training with Non-expert Human Demonstration for Deep Reinforcement Learning","date":"2018-12-21","arxiv_id":"1812.08904","repositories_listed":1,"syntology":null},{"url":"/paper/td-regularized-actor-critic-methods","slug":"td-regularized-actor-critic-methods","title":"TD-Regularized Actor-Critic Methods","date":"2018-12-19","arxiv_id":"1812.08288","repositories_listed":1,"syntology":null},{"url":"/paper/information-directed-exploration-for-deep","slug":"information-directed-exploration-for-deep","title":"Information-Directed Exploration for Deep Reinforcement Learning","date":"2018-12-18","arxiv_id":"1812.07544","repositories_listed":1,"syntology":null},{"url":"/paper/an-atari-model-zoo-for-analyzing-visualizing","slug":"an-atari-model-zoo-for-analyzing-visualizing","title":"An Atari Model Zoo for Analyzing, Visualizing, and Comparing Deep Reinforcement Learning Agents","date":"2018-12-17","arxiv_id":"1812.07069","repositories_listed":1,"syntology":null},{"url":"/paper/residual-policy-learning","slug":"residual-policy-learning","title":"Residual Policy Learning","date":"2018-12-15","arxiv_id":"1812.06298","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/residual-policy-learning#ran","syntology_url":"https://syntology.ai/paper/1812.06298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.06298"}},"official":null}},{"url":"/paper/simulation-to-scaled-city-zero-shot-policy","slug":"simulation-to-scaled-city-zero-shot-policy","title":"Simulation to Scaled City: Zero-Shot Policy Transfer for Traffic Control via Autonomous Vehicles","date":"2018-12-14","arxiv_id":"1812.06120","repositories_listed":1,"syntology":null},{"url":"/paper/exploration-conscious-reinforcement-learning","slug":"exploration-conscious-reinforcement-learning","title":"Exploration Conscious Reinforcement Learning Revisited","date":"2018-12-13","arxiv_id":"1812.05551","repositories_listed":1,"syntology":null},{"url":"/paper/irlas-inverse-reinforcement-learning-for","slug":"irlas-inverse-reinforcement-learning-for","title":"IRLAS: Inverse Reinforcement Learning for Architecture Search","date":"2018-12-13","arxiv_id":"1812.05285","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-on-graphs-a-survey","slug":"deep-learning-on-graphs-a-survey","title":"Deep Learning on Graphs: A Survey","date":"2018-12-11","arxiv_id":"1812.04202","repositories_listed":1,"syntology":null},{"url":"/paper/dialogue-generation-from-imitation-learning","slug":"dialogue-generation-from-imitation-learning","title":"Dialogue Generation: From Imitation Learning to Inverse Reinforcement Learning","date":"2018-12-09","arxiv_id":"1812.03509","repositories_listed":1,"syntology":null},{"url":"/paper/weighted-risk-minimization-deep-learning","slug":"weighted-risk-minimization-deep-learning","title":"What is the Effect of Importance Weighting in Deep Learning?","date":"2018-12-08","arxiv_id":"1812.03372","repositories_listed":1,"syntology":null},{"url":"/paper/pseudo-rehearsal-achieving-deep-reinforcement","slug":"pseudo-rehearsal-achieving-deep-reinforcement","title":"Pseudo-Rehearsal: Achieving Deep Reinforcement Learning without Catastrophic Forgetting","date":"2018-12-06","arxiv_id":"1812.02464","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-generalization-in-reinforcement","slug":"quantifying-generalization-in-reinforcement","title":"Quantifying Generalization in Reinforcement Learning","date":"2018-12-06","arxiv_id":"1812.02341","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-generalization-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1812.02341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.02341"}},"official":{"repos":["openai/coinrun"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-auxiliary-losses-using-gradient","slug":"adapting-auxiliary-losses-using-gradient","title":"Adapting Auxiliary Losses Using Gradient Similarity","date":"2018-12-05","arxiv_id":"1812.02224","repositories_listed":1,"syntology":null}],"record_sha256":"09dd529c438c23d48ed9d9999a1d7fdcb0964c421d761638a100eae3aee43c2e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}