{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/ran/13","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":13,"pages_in_order":15,"rows_per_page":100,"rows":[1201,1300],"of":1416,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1/papers/ran/1","prev":"/task/reinforcement-learning-1/papers/ran/12","next":"/task/reinforcement-learning-1/papers/ran/14","papers":[{"url":"/paper/learning-phase-competition-for-traffic-signal","slug":"learning-phase-competition-for-traffic-signal","title":"Learning Phase Competition for Traffic Signal Control","date":"2019-05-12","arxiv_id":"1905.04722","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-phase-competition-for-traffic-signal#ran","syntology_url":"https://syntology.ai/paper/1905.04722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04722"}},"official":null}},{"url":"/paper/dimension-wise-importance-sampling-weight","slug":"dimension-wise-importance-sampling-weight","title":"Dimension-Wise Importance Sampling Weight Clipping for Sample-Efficient Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.02363","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":12,"n_pointer_only":18,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dimension-wise-importance-sampling-weight#ran","syntology_url":"https://syntology.ai/paper/1905.02363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.02363"}},"official":{"repos":["seungyulhan/disc"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-control-in-metric-space-with","slug":"learning-to-control-in-metric-space-with","title":"Learning to Control in Metric Space with Optimal Regret","date":"2019-05-05","arxiv_id":"1905.01576","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-in-metric-space-with#ran","syntology_url":"https://syntology.ai/paper/1905.01576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.01576"}},"official":null}},{"url":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1905.00976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00976"}},"official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-gan-net-a-reinforcement-learning-agent","slug":"rl-gan-net-a-reinforcement-learning-agent","title":"RL-GAN-Net: A Reinforcement Learning Agent Controlled GAN Network for Real-Time Point Cloud Shape Completion","date":"2019-04-28","arxiv_id":"1904.12304","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rl-gan-net-a-reinforcement-learning-agent#ran","syntology_url":"https://syntology.ai/paper/1904.12304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.12304"}},"official":null}},{"url":"/paper/deep-neuroevolution-of-recurrent-and-discrete","slug":"deep-neuroevolution-of-recurrent-and-discrete","title":"Deep Neuroevolution of Recurrent and Discrete World Models","date":"2019-04-28","arxiv_id":"1906.08857","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-neuroevolution-of-recurrent-and-discrete#ran","syntology_url":"https://syntology.ai/paper/1906.08857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.08857"}},"official":{"repos":["sebastianrisi/ga-world-models"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/neural-logic-reinforcement-learning","slug":"neural-logic-reinforcement-learning","title":"Neural Logic Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.10729","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neural-logic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1904.10729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10729"}},"official":{"repos":["ZhengyaoJiang/NLRL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/graphnas-graph-neural-architecture-search","slug":"graphnas-graph-neural-architecture-search","title":"GraphNAS: Graph Neural Architecture Search with Reinforcement Learning","date":"2019-04-22","arxiv_id":"1904.09981","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/graphnas-graph-neural-architecture-search#ran","syntology_url":"https://syntology.ai/paper/1904.09981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09981"}},"official":{"repos":["GraphNAS/GraphNAS-simple"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rogue-gym-a-new-challenge-for-generalization","slug":"rogue-gym-a-new-challenge-for-generalization","title":"Rogue-Gym: A New Challenge for Generalization in Reinforcement Learning","date":"2019-04-17","arxiv_id":"1904.08129","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rogue-gym-a-new-challenge-for-generalization#ran","syntology_url":"https://syntology.ai/paper/1904.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08129"}},"official":{"repos":["kngwyu/rogue-gym","kngwyu/rogue-gym-agents-cog19"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hitchhikers-guide-to-statistical","slug":"a-hitchhikers-guide-to-statistical","title":"A Hitchhiker's Guide to Statistical Comparisons of Reinforcement Learning Algorithms","date":"2019-04-15","arxiv_id":"1904.06979","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hitchhikers-guide-to-statistical#ran","syntology_url":"https://syntology.ai/paper/1904.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06979"}},"official":{"repos":["flowersteam/rl_stats","ccolas/rl_stats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extrapolating-beyond-suboptimal","slug":"extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extrapolating-beyond-suboptimal#ran","syntology_url":"https://syntology.ai/paper/1904.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06387"}},"official":{"repos":["hiwonjoon/ICML2019-TREX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-reinforcement-learning-via","slug":"interpretable-reinforcement-learning-via","title":"Optimization Methods for Interpretable Differentiable Decision Trees in Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09338","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interpretable-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/1903.09338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.09338"}},"official":null}},{"url":"/paper/efficient-off-policy-meta-reinforcement","slug":"efficient-off-policy-meta-reinforcement","title":"Efficient Off-Policy Meta-Reinforcement Learning via Probabilistic Context Variables","date":"2019-03-19","arxiv_id":"1903.08254","repositories_listed":7,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-off-policy-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1903.08254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.08254"}},"official":{"repos":["katerakelly/oyster"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/coacor-code-annotation-for-code-retrieval","slug":"coacor-code-annotation-for-code-retrieval","title":"CoaCor: Code Annotation for Code Retrieval with Reinforcement Learning","date":"2019-03-13","arxiv_id":"1904.00720","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/coacor-code-annotation-for-code-retrieval#ran","syntology_url":"https://syntology.ai/paper/1904.00720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.00720"}},"official":{"repos":["LittleYUYU/CoaCor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-reinforcement-learning-with-expert","slug":"hybrid-reinforcement-learning-with-expert","title":"Hybrid Reinforcement Learning with Expert State Sequences","date":"2019-03-11","arxiv_id":"1903.04110","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-reinforcement-learning-with-expert#ran","syntology_url":"https://syntology.ai/paper/1903.04110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04110"}},"official":{"repos":["XiaoxiaoGuo/tensor4rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stroke-based-artistic-rendering-agent-with","slug":"stroke-based-artistic-rendering-agent-with","title":"Learning to Paint With Model-based Deep Reinforcement Learning","date":"2019-03-11","arxiv_id":"1903.04411","repositories_listed":6,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/stroke-based-artistic-rendering-agent-with#ran","syntology_url":"https://syntology.ai/paper/1903.04411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04411"}},"official":{"repos":["hzwer/ICCV2019-LearningToPaint"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-2","slug":"multi-agent-deep-reinforcement-learning-for-2","title":"Multi-Agent Deep Reinforcement Learning for Large-scale Traffic Signal Control","date":"2019-03-11","arxiv_id":"1903.04527","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-deep-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/1903.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04527"}},"official":{"repos":["cts198859/deeprl_signal_control"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-power-system-emergency-control-using","slug":"adaptive-power-system-emergency-control-using","title":"Adaptive Power System Emergency Control using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03712","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-power-system-emergency-control-using#ran","syntology_url":"https://syntology.ai/paper/1903.03712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03712"}},"official":{"repos":["RLGC-Project/RLGC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minatar-an-atari-inspired-testbed-for-more","slug":"minatar-an-atari-inspired-testbed-for-more","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","date":"2019-03-07","arxiv_id":"1903.03176","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/minatar-an-atari-inspired-testbed-for-more#ran","syntology_url":"https://syntology.ai/paper/1903.03176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03176"}},"official":{"repos":["kenjyoung/MinAtar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/using-natural-language-for-reward-shaping-in","slug":"using-natural-language-for-reward-shaping-in","title":"Using Natural Language for Reward Shaping in Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.02020","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/using-natural-language-for-reward-shaping-in#ran","syntology_url":"https://syntology.ai/paper/1903.02020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.02020"}},"official":null}},{"url":"/paper/model-primitive-hierarchical-lifelong","slug":"model-primitive-hierarchical-lifelong","title":"Model Primitive Hierarchical Lifelong Reinforcement Learning","date":"2019-03-04","arxiv_id":"1903.01567","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-primitive-hierarchical-lifelong#ran","syntology_url":"https://syntology.ai/paper/1903.01567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.01567"}},"official":{"repos":["sisl/MPHRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-reinforcement-learning-for-atari","slug":"model-based-reinforcement-learning-for-atari","title":"Model-Based Reinforcement Learning for Atari","date":"2019-03-01","arxiv_id":"1903.00374","repositories_listed":2,"syntology":{"n":21,"n_ran":16,"n_constructed":8,"n_ran_checked":10,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"16 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-atari#ran","syntology_url":"https://syntology.ai/paper/1903.00374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.00374"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/diagnosing-bottlenecks-in-deep-q-learning","slug":"diagnosing-bottlenecks-in-deep-q-learning","title":"Diagnosing Bottlenecks in Deep Q-learning Algorithms","date":"2019-02-26","arxiv_id":"1902.10250","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diagnosing-bottlenecks-in-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1902.10250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.10250"}},"official":null}},{"url":"/paper/a-general-framework-for-structured-learning","slug":"a-general-framework-for-structured-learning","title":"A General Framework for Structured Learning of Mechanical Systems","date":"2019-02-22","arxiv_id":"1902.08705","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-general-framework-for-structured-learning#ran","syntology_url":"https://syntology.ai/paper/1902.08705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.08705"}},"official":{"repos":["sisl/mechamodlearn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/190504100","slug":"190504100","title":"Deep Reinforcement Learning using Genetic Algorithm for Parameter Optimization","date":"2019-02-19","arxiv_id":"1905.04100","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/190504100#ran","syntology_url":"https://syntology.ai/paper/1905.04100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04100"}},"official":{"repos":["aralab-unr/ReinforcementLearningWithGA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/preferences-implicit-in-the-state-of-the","slug":"preferences-implicit-in-the-state-of-the","title":"Preferences Implicit in the State of the World","date":"2019-02-12","arxiv_id":"1902.04198","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preferences-implicit-in-the-state-of-the#ran","syntology_url":"https://syntology.ai/paper/1902.04198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.04198"}},"official":{"repos":["HumanCompatibleAI/rlsp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-starcraft-multi-agent-challenge","slug":"the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/the-starcraft-multi-agent-challenge#ran","syntology_url":"https://syntology.ai/paper/1902.04043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.04043"}},"official":{"repos":["oxwhirl/pymarl","oxwhirl/smac"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/artificial-intelligence-for-prosthetics","slug":"artificial-intelligence-for-prosthetics","title":"Artificial Intelligence for Prosthetics - challenge solutions","date":"2019-02-07","arxiv_id":"1902.02441","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/artificial-intelligence-for-prosthetics#ran","syntology_url":"https://syntology.ai/paper/1902.02441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.02441"}},"official":{"repos":["iasawseen/MultiServerRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-consolidation-for-continual","slug":"policy-consolidation-for-continual","title":"Policy Consolidation for Continual Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00255","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-consolidation-for-continual#ran","syntology_url":"https://syntology.ai/paper/1902.00255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00255"}},"official":null}},{"url":"/paper/action-robust-reinforcement-learning-and","slug":"action-robust-reinforcement-learning-and","title":"Action Robust Reinforcement Learning and Applications in Continuous Control","date":"2019-01-26","arxiv_id":"1901.09184","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-robust-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/1901.09184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09184"}},"official":{"repos":["tesslerc/ActionRobustRL","icml2019-anonymous-author/Action-Robust-Reinforcement-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-linguistic-phenomena-in-multi-agent","slug":"emergent-linguistic-phenomena-in-multi-agent","title":"Emergent Linguistic Phenomena in Multi-Agent Communication Games","date":"2019-01-25","arxiv_id":"1901.08706","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/emergent-linguistic-phenomena-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1901.08706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08706"}},"official":{"repos":["lgraesser/MultimodalGame"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-agile-and-dynamic-motor-skills-for","slug":"learning-agile-and-dynamic-motor-skills-for","title":"Learning agile and dynamic motor skills for legged robots","date":"2019-01-24","arxiv_id":"1901.08652","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-agile-and-dynamic-motor-skills-for#ran","syntology_url":"https://syntology.ai/paper/1901.08652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08652"}},"official":{"repos":["junja94/anymal_science_robotics_supplementary"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-multi-agent-reinforcement-learning-in","slug":"the-multi-agent-reinforcement-learning-in","title":"The Multi-Agent Reinforcement Learning in MalmÖ (MARLÖ) Competition","date":"2019-01-23","arxiv_id":"1901.08129","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-multi-agent-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/1901.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08129"}},"official":null}},{"url":"/paper/causal-reasoning-from-meta-reinforcement","slug":"causal-reasoning-from-meta-reinforcement","title":"Causal Reasoning from Meta-reinforcement Learning","date":"2019-01-23","arxiv_id":"1901.08162","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causal-reasoning-from-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.08162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08162"}},"official":null}},{"url":"/paper/fast-accurate-and-lightweight-super","slug":"fast-accurate-and-lightweight-super","title":"Fast, Accurate and Lightweight Super-Resolution with Neural Architecture Search","date":"2019-01-22","arxiv_id":"1901.07261","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-accurate-and-lightweight-super#ran","syntology_url":"https://syntology.ai/paper/1901.07261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.07261"}},"official":{"repos":["falsr/FALSR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/improving-coordination-in-multi-agent-deep","slug":"improving-coordination-in-multi-agent-deep","title":"Improving Coordination in Small-Scale Multi-Agent Deep Reinforcement Learning through Memory-driven Communication","date":"2019-01-12","arxiv_id":"1901.03887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-coordination-in-multi-agent-deep#ran","syntology_url":"https://syntology.ai/paper/1901.03887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.03887"}},"official":null}},{"url":"/paper/risk-aware-active-inverse-reinforcement","slug":"risk-aware-active-inverse-reinforcement","title":"Risk-Aware Active Inverse Reinforcement Learning","date":"2019-01-08","arxiv_id":"1901.02161","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/risk-aware-active-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.02161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.02161"}},"official":{"repos":["Pearl-UTexas/ActiveVaR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-imbalanced#ran","syntology_url":"https://syntology.ai/paper/1901.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01379"}},"official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/dopamine-a-research-framework-for-deep","slug":"dopamine-a-research-framework-for-deep","title":"Dopamine: A Research Framework for Deep Reinforcement Learning","date":"2018-12-14","arxiv_id":"1812.06110","repositories_listed":13,"syntology":{"n":16,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/dopamine-a-research-framework-for-deep#ran","syntology_url":"https://syntology.ai/paper/1812.06110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.06110"}},"official":{"repos":["google/dopamine"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/soft-actor-critic-algorithms-and-applications","slug":"soft-actor-critic-algorithms-and-applications","title":"Soft Actor-Critic Algorithms and Applications","date":"2018-12-13","arxiv_id":"1812.05905","repositories_listed":52,"syntology":{"n":40,"n_ran":38,"n_constructed":0,"n_ran_checked":35,"n_instrument":3,"n_unverified":2,"n_honours":4,"n_violates":1,"n_no_contract":30,"n_pointer_only":5,"phrase":"38 ran (of which 0 constructed an object rather than computing a result; 35 with no instrument failure: 4 honoured, 1 violated, 30 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soft-actor-critic-algorithms-and-applications#ran","syntology_url":"https://syntology.ai/paper/1812.05905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.05905"}},"official":{"repos":["rail-berkeley/softlearning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/off-policy-deep-reinforcement-learning","slug":"off-policy-deep-reinforcement-learning","title":"Off-Policy Deep Reinforcement Learning without Exploration","date":"2018-12-07","arxiv_id":"1812.02900","repositories_listed":10,"syntology":{"n":14,"n_ran":14,"n_constructed":12,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":13,"n_pointer_only":9,"phrase":"14 ran (of which 12 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-policy-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1812.02900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.02900"}},"official":{"repos":["sfujim/BCQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/quantifying-generalization-in-reinforcement","slug":"quantifying-generalization-in-reinforcement","title":"Quantifying Generalization in Reinforcement Learning","date":"2018-12-06","arxiv_id":"1812.02341","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-generalization-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1812.02341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.02341"}},"official":{"repos":["openai/coinrun"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/deep-multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1811.12557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.12557"}},"official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-impact-of-entropy-on-policy","slug":"understanding-the-impact-of-entropy-on-policy","title":"Understanding the impact of entropy on policy optimization","date":"2018-11-27","arxiv_id":"1811.11214","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/understanding-the-impact-of-entropy-on-policy#ran","syntology_url":"https://syntology.ai/paper/1811.11214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.11214"}},"official":{"repos":["zafarali/emdp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-uplift-modeling","slug":"reinforcement-learning-for-uplift-modeling","title":"Reinforcement Learning for Uplift Modeling","date":"2018-11-26","arxiv_id":"1811.10158","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-uplift-modeling#ran","syntology_url":"https://syntology.ai/paper/1811.10158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.10158"}},"official":null}},{"url":"/paper/improving-automatic-source-code-summarization","slug":"improving-automatic-source-code-summarization","title":"Improving Automatic Source Code Summarization via Deep Reinforcement Learning","date":"2018-11-17","arxiv_id":"1811.07234","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-automatic-source-code-summarization#ran","syntology_url":"https://syntology.ai/paper/1811.07234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.07234"}},"official":null}},{"url":"/paper/quota-the-quantile-option-architecture-for","slug":"quota-the-quantile-option-architecture-for","title":"QUOTA: The Quantile Option Architecture for Reinforcement Learning","date":"2018-11-05","arxiv_id":"1811.02073","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quota-the-quantile-option-architecture-for#ran","syntology_url":"https://syntology.ai/paper/1811.02073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.02073"}},"official":{"repos":["ShangtongZhang/DeepRL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/exploration-by-random-network-distillation","slug":"exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","arxiv_id":"1810.12894","repositories_listed":22,"syntology":{"n":43,"n_ran":31,"n_constructed":12,"n_ran_checked":24,"n_instrument":7,"n_unverified":12,"n_honours":2,"n_violates":1,"n_no_contract":21,"n_pointer_only":16,"phrase":"31 ran (of which 12 constructed an object rather than computing a result; 24 with no instrument failure: 2 honoured, 1 violated, 21 with no contract checked; 7 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/exploration-by-random-network-distillation#ran","syntology_url":"https://syntology.ai/paper/1810.12894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12894"}},"official":{"repos":["openai/random-network-distillation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/model-based-active-exploration","slug":"model-based-active-exploration","title":"Model-Based Active Exploration","date":"2018-10-29","arxiv_id":"1810.12162","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-active-exploration#ran","syntology_url":"https://syntology.ai/paper/1810.12162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12162"}},"official":{"repos":["nnaisense/max"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/assessing-generalization-in-deep","slug":"assessing-generalization-in-deep","title":"Assessing Generalization in Deep Reinforcement Learning","date":"2018-10-29","arxiv_id":"1810.12282","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assessing-generalization-in-deep#ran","syntology_url":"https://syntology.ai/paper/1810.12282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12282"}},"official":{"repos":["sunblaze-ucb/rl-generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-common-knowledge-reinforcement","slug":"multi-agent-common-knowledge-reinforcement","title":"Multi-Agent Common Knowledge Reinforcement Learning","date":"2018-10-27","arxiv_id":"1810.11702","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-common-knowledge-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1810.11702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11702"}},"official":{"repos":["schroederdewitt/mackrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-modular-control-for-embodied-question","slug":"neural-modular-control-for-embodied-question","title":"Neural Modular Control for Embodied Question Answering","date":"2018-10-26","arxiv_id":"1810.11181","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/neural-modular-control-for-embodied-question#ran","syntology_url":"https://syntology.ai/paper/1810.11181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11181"}},"official":null}},{"url":"/paper/making-sense-of-vision-and-touch-self","slug":"making-sense-of-vision-and-touch-self","title":"Making Sense of Vision and Touch: Self-Supervised Learning of Multimodal Representations for Contact-Rich Tasks","date":"2018-10-24","arxiv_id":"1810.10191","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/making-sense-of-vision-and-touch-self#ran","syntology_url":"https://syntology.ai/paper/1810.10191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.10191"}},"official":{"repos":["stanford-iprl-lab/multimodal_representation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-convolutional-reinforcement-learning","slug":"graph-convolutional-reinforcement-learning","title":"Graph Convolutional Reinforcement Learning","date":"2018-10-22","arxiv_id":"1810.09202","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-convolutional-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.09202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09202"}},"official":{"repos":["PKU-AI-Edge/DGN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fast-deep-reinforcement-learning-using-online","slug":"fast-deep-reinforcement-learning-using-online","title":"Fast deep reinforcement learning using online adjustments from the past","date":"2018-10-18","arxiv_id":"1810.08163","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-deep-reinforcement-learning-using-online#ran","syntology_url":"https://syntology.ai/paper/1810.08163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.08163"}},"official":null}},{"url":"/paper/successor-uncertainties-exploration-and","slug":"successor-uncertainties-exploration-and","title":"Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","date":"2018-10-15","arxiv_id":"1810.06530","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/successor-uncertainties-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1810.06530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06530"}},"official":null}},{"url":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parametrized-deep-q-networks-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06394"}},"official":null}},{"url":"/paper/fast-context-adaptation-via-meta-learning","slug":"fast-context-adaptation-via-meta-learning","title":"Fast Context Adaptation via Meta-Learning","date":"2018-10-08","arxiv_id":"1810.03642","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-context-adaptation-via-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1810.03642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.03642"}},"official":{"repos":["lmzintgraf/cavia"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/actor-attention-critic-for-multi-agent","slug":"actor-attention-critic-for-multi-agent","title":"Actor-Attention-Critic for Multi-Agent Reinforcement Learning","date":"2018-10-05","arxiv_id":"1810.02912","repositories_listed":3,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-attention-critic-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1810.02912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02912"}},"official":{"repos":["shariqiqbal2810/MAAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-scheduling-algorithms-for-data","slug":"learning-scheduling-algorithms-for-data","title":"Learning Scheduling Algorithms for Data Processing Clusters","date":"2018-10-03","arxiv_id":"1810.01963","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-scheduling-algorithms-for-data#ran","syntology_url":"https://syntology.ai/paper/1810.01963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01963"}},"official":{"repos":["hongzimao/decima-sim"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-with-perturbed-rewards","slug":"reinforcement-learning-with-perturbed-rewards","title":"Reinforcement Learning with Perturbed Rewards","date":"2018-10-02","arxiv_id":"1810.01032","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-perturbed-rewards#ran","syntology_url":"https://syntology.ai/paper/1810.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01032"}},"official":{"repos":["wangjksjtu/rl-perturbed-reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emi-exploration-with-mutual-information","slug":"emi-exploration-with-mutual-information","title":"EMI: Exploration with Mutual Information","date":"2018-10-02","arxiv_id":"1810.01176","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emi-exploration-with-mutual-information#ran","syntology_url":"https://syntology.ai/paper/1810.01176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01176"}},"official":{"repos":["snu-mllab/EMI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-statistical-mechanics-using","slug":"solving-statistical-mechanics-using","title":"Solving Statistical Mechanics Using Variational Autoregressive Networks","date":"2018-09-27","arxiv_id":"1809.10606","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/solving-statistical-mechanics-using#ran","syntology_url":"https://syntology.ai/paper/1809.10606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.10606"}},"official":{"repos":["wangleiphy/VAN.jl","wdphy16/stat-mech-van"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crowd-robot-interaction-crowd-aware-robot","slug":"crowd-robot-interaction-crowd-aware-robot","title":"Crowd-Robot Interaction: Crowd-aware Robot Navigation with Attention-based Deep Reinforcement Learning","date":"2018-09-24","arxiv_id":"1809.08835","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crowd-robot-interaction-crowd-aware-robot#ran","syntology_url":"https://syntology.ai/paper/1809.08835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.08835"}},"official":{"repos":["vita-epfl/CrowdNav","vita-epfl/DyNav"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/benchmarking-reinforcement-learning","slug":"benchmarking-reinforcement-learning","title":"Benchmarking Reinforcement Learning Algorithms on Real-World Robots","date":"2018-09-20","arxiv_id":"1809.07731","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1809.07731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.07731"}},"official":{"repos":["kindredresearch/SenseAct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-across-multi-objective-reward","slug":"generalizing-across-multi-objective-reward","title":"Generalizing Across Multi-Objective Reward Functions in Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06364","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generalizing-across-multi-objective-reward#ran","syntology_url":"https://syntology.ai/paper/1809.06364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06364"}},"official":null}},{"url":"/paper/towards-better-interpretability-in-deep-q","slug":"towards-better-interpretability-in-deep-q","title":"Towards Better Interpretability in Deep Q-Networks","date":"2018-09-15","arxiv_id":"1809.05630","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-better-interpretability-in-deep-q#ran","syntology_url":"https://syntology.ai/paper/1809.05630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05630"}},"official":null}},{"url":"/paper/cm3-cooperative-multi-goal-multi-stage-multi","slug":"cm3-cooperative-multi-goal-multi-stage-multi","title":"CM3: Cooperative Multi-goal Multi-stage Multi-agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05188","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cm3-cooperative-multi-goal-multi-stage-multi#ran","syntology_url":"https://syntology.ai/paper/1809.05188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05188"}},"official":{"repos":["011235813/cm3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with","slug":"multi-task-deep-reinforcement-learning-with","title":"Multi-task Deep Reinforcement Learning with PopArt","date":"2018-09-12","arxiv_id":"1809.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1809.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.04474"}},"official":null}},{"url":"/paper/improving-optimization-bounds-using-machine","slug":"improving-optimization-bounds-using-machine","title":"Improving Optimization Bounds using Machine Learning: Decision Diagrams meet Deep Reinforcement Learning","date":"2018-09-10","arxiv_id":"1809.03359","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-optimization-bounds-using-machine#ran","syntology_url":"https://syntology.ai/paper/1809.03359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.03359"}},"official":{"repos":["qcappart/learning-DD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-hop-knowledge-graph-reasoning-with","slug":"multi-hop-knowledge-graph-reasoning-with","title":"Multi-Hop Knowledge Graph Reasoning with Reward Shaping","date":"2018-08-31","arxiv_id":"1808.10568","repositories_listed":3,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-hop-knowledge-graph-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/1808.10568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.10568"}},"official":{"repos":["salesforce/MultiHopKG"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/decoupling-strategy-and-generation-in","slug":"decoupling-strategy-and-generation-in","title":"Decoupling Strategy and Generation in Negotiation Dialogues","date":"2018-08-29","arxiv_id":"1808.09637","repositories_listed":3,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/decoupling-strategy-and-generation-in#ran","syntology_url":"https://syntology.ai/paper/1808.09637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09637"}},"official":{"repos":["worksheets.codalab.org/worksheets/0x453913e76b65495d8b9730d41c7e0a0c"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/april-interactively-learning-to-summarise-by","slug":"april-interactively-learning-to-summarise-by","title":"APRIL: Interactively Learning to Summarise by Combining Active Preference Learning and Reinforcement Learning","date":"2018-08-29","arxiv_id":"1808.09658","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/april-interactively-learning-to-summarise-by#ran","syntology_url":"https://syntology.ai/paper/1808.09658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09658"}},"official":{"repos":["UKPLab/emnlp2018-april"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-study-of-reinforcement-learning-for-neural","slug":"a-study-of-reinforcement-learning-for-neural","title":"A Study of Reinforcement Learning for Neural Machine Translation","date":"2018-08-27","arxiv_id":"1808.08866","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-study-of-reinforcement-learning-for-neural#ran","syntology_url":"https://syntology.ai/paper/1808.08866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.08866"}},"official":{"repos":["apeterswu/RL4NMT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recogym-a-reinforcement-learning-environment","slug":"recogym-a-reinforcement-learning-environment","title":"RecoGym: A Reinforcement Learning Environment for the problem of Product Recommendation in Online Advertising","date":"2018-08-02","arxiv_id":"1808.00720","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recogym-a-reinforcement-learning-environment#ran","syntology_url":"https://syntology.ai/paper/1808.00720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.00720"}},"official":{"repos":["criteo-research/reco-gym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/count-based-exploration-with-the-successor","slug":"count-based-exploration-with-the-successor","title":"Count-Based Exploration with the Successor Representation","date":"2018-07-31","arxiv_id":"1807.11622","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/count-based-exploration-with-the-successor#ran","syntology_url":"https://syntology.ai/paper/1807.11622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.11622"}},"official":{"repos":["mcmachado/count_based_exploration_sr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-generative-adversarial-imitation","slug":"multi-agent-generative-adversarial-imitation","title":"Multi-Agent Generative Adversarial Imitation Learning","date":"2018-07-26","arxiv_id":"1807.09936","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-generative-adversarial-imitation#ran","syntology_url":"https://syntology.ai/paper/1807.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.09936"}},"official":null}},{"url":"/paper/learning-to-listen-read-and-follow-score","slug":"learning-to-listen-read-and-follow-score","title":"Learning to Listen, Read, and Follow: Score Following as a Reinforcement Learning Game","date":"2018-07-17","arxiv_id":"1807.06391","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-to-listen-read-and-follow-score#ran","syntology_url":"https://syntology.ai/paper/1807.06391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.06391"}},"official":{"repos":["CPJKU/score_following_game"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/remember-and-forget-for-experience-replay","slug":"remember-and-forget-for-experience-replay","title":"Remember and Forget for Experience Replay","date":"2018-07-16","arxiv_id":"1807.05827","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/remember-and-forget-for-experience-replay#ran","syntology_url":"https://syntology.ai/paper/1807.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.05827"}},"official":{"repos":["cselab/smarties"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/algorithmic-framework-for-model-based-deep","slug":"algorithmic-framework-for-model-based-deep","title":"Algorithmic Framework for Model-based Deep Reinforcement Learning with Theoretical Guarantees","date":"2018-07-10","arxiv_id":"1807.03858","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/algorithmic-framework-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1807.03858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03858"}},"official":{"repos":["roosephu/slbo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-drive-in-a-day","slug":"learning-to-drive-in-a-day","title":"Learning to Drive in a Day","date":"2018-07-01","arxiv_id":"1807.00412","repositories_listed":8,"syntology":{"n":22,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-to-drive-in-a-day#ran","syntology_url":"https://syntology.ai/paper/1807.00412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.00412"}},"official":null}},{"url":"/paper/darts-differentiable-architecture-search","slug":"darts-differentiable-architecture-search","title":"DARTS: Differentiable Architecture Search","date":"2018-06-24","arxiv_id":"1806.09055","repositories_listed":59,"syntology":{"n":156,"n_ran":74,"n_constructed":30,"n_ran_checked":60,"n_instrument":14,"n_unverified":82,"n_honours":9,"n_violates":1,"n_no_contract":50,"n_pointer_only":54,"phrase":"74 ran (of which 30 constructed an object rather than computing a result; 60 with no instrument failure: 9 honoured, 1 violated, 50 with no contract checked; 14 where Syntology's instrument failed) · 82 unverified","sample_list":"/paper/darts-differentiable-architecture-search#ran","syntology_url":"https://syntology.ai/paper/1806.09055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.09055"}},"official":{"repos":["quark0/darts"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/sim-to-real-reinforcement-learning-for","slug":"sim-to-real-reinforcement-learning-for","title":"Sim-to-Real Reinforcement Learning for Deformable Object Manipulation","date":"2018-06-20","arxiv_id":"1806.07851","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sim-to-real-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.07851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07851"}},"official":{"repos":["JanMatas/Rainbow_ddpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rudder-return-decomposition-for-delayed","slug":"rudder-return-decomposition-for-delayed","title":"RUDDER: Return Decomposition for Delayed Rewards","date":"2018-06-20","arxiv_id":"1806.07857","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rudder-return-decomposition-for-delayed#ran","syntology_url":"https://syntology.ai/paper/1806.07857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07857"}},"official":{"repos":["ml-jku/baselines-rudder","ml-jku/rudder"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/barc-backward-reachability-curriculum-for","slug":"barc-backward-reachability-curriculum-for","title":"BaRC: Backward Reachability Curriculum for Robotic Reinforcement Learning","date":"2018-06-16","arxiv_id":"1806.06161","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/barc-backward-reachability-curriculum-for#ran","syntology_url":"https://syntology.ai/paper/1806.06161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.06161"}},"official":{"repos":["StanfordASL/BaRC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/implicit-quantile-networks-for-distributional","slug":"implicit-quantile-networks-for-distributional","title":"Implicit Quantile Networks for Distributional Reinforcement Learning","date":"2018-06-14","arxiv_id":"1806.06923","repositories_listed":19,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-quantile-networks-for-distributional#ran","syntology_url":"https://syntology.ai/paper/1806.06923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.06923"}},"official":null}},{"url":"/paper/structured-variational-learning-of-bayesian","slug":"structured-variational-learning-of-bayesian","title":"Structured Variational Learning of Bayesian Neural Networks with Horseshoe Priors","date":"2018-06-13","arxiv_id":"1806.05975","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-variational-learning-of-bayesian#ran","syntology_url":"https://syntology.ai/paper/1806.05975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.05975"}},"official":null}},{"url":"/paper/the-potential-of-the-return-distribution-for","slug":"the-potential-of-the-return-distribution-for","title":"The Potential of the Return Distribution for Exploration in RL","date":"2018-06-11","arxiv_id":"1806.04242","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-potential-of-the-return-distribution-for#ran","syntology_url":"https://syntology.ai/paper/1806.04242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.04242"}},"official":{"repos":["tmoer/return_distribution_exploration"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/randomized-prior-functions-for-deep","slug":"randomized-prior-functions-for-deep","title":"Randomized Prior Functions for Deep Reinforcement Learning","date":"2018-06-08","arxiv_id":"1806.03335","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/randomized-prior-functions-for-deep#ran","syntology_url":"https://syntology.ai/paper/1806.03335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.03335"}},"official":null}},{"url":"/paper/deep-variational-reinforcement-learning-for","slug":"deep-variational-reinforcement-learning-for","title":"Deep Variational Reinforcement Learning for POMDPs","date":"2018-06-06","arxiv_id":"1806.02426","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-variational-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.02426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02426"}},"official":{"repos":["maximilianigl/DVRL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-general-video","slug":"deep-reinforcement-learning-for-general-video","title":"Deep Reinforcement Learning for General Video Game AI","date":"2018-06-06","arxiv_id":"1806.02448","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-general-video#ran","syntology_url":"https://syntology.ai/paper/1806.02448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02448"}},"official":{"repos":["rubenrtorrado/GVGAI_GYM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/relational-deep-reinforcement-learning","slug":"relational-deep-reinforcement-learning","title":"Relational Deep Reinforcement Learning","date":"2018-06-05","arxiv_id":"1806.01830","repositories_listed":7,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/relational-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1806.01830","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01830"}},"official":null}},{"url":"/paper/challenges-in-high-dimensional-reinforcement","slug":"challenges-in-high-dimensional-reinforcement","title":"Challenges in High-dimensional Reinforcement Learning with Evolution Strategies","date":"2018-06-04","arxiv_id":"1806.01224","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/challenges-in-high-dimensional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1806.01224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01224"}},"official":{"repos":["NiMlr/High-Dim-ES-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-in-a-handful-of","slug":"deep-reinforcement-learning-in-a-handful-of","title":"Deep Reinforcement Learning in a Handful of Trials using Probabilistic Dynamics Models","date":"2018-05-30","arxiv_id":"1805.12114","repositories_listed":9,"syntology":{"n":19,"n_ran":12,"n_constructed":5,"n_ran_checked":7,"n_instrument":5,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":17,"phrase":"12 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/deep-reinforcement-learning-in-a-handful-of#ran","syntology_url":"https://syntology.ai/paper/1805.12114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.12114"}},"official":{"repos":["kchua/handful-of-trials"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reliability-and-learnability-of-human-bandit","slug":"reliability-and-learnability-of-human-bandit","title":"Reliability and Learnability of Human Bandit Feedback for Sequence-to-Sequence Reinforcement Learning","date":"2018-05-27","arxiv_id":"1805.10627","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reliability-and-learnability-of-human-bandit#ran","syntology_url":"https://syntology.ai/paper/1805.10627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.10627"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-sequence-to","slug":"deep-reinforcement-learning-for-sequence-to","title":"Deep Reinforcement Learning For Sequence to Sequence Models","date":"2018-05-24","arxiv_id":"1805.09461","repositories_listed":3,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sequence-to#ran","syntology_url":"https://syntology.ai/paper/1805.09461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09461"}},"official":{"repos":["yaserkl/RLSeq2Seq"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a0c-alpha-zero-in-continuous-action-space","slug":"a0c-alpha-zero-in-continuous-action-space","title":"A0C: Alpha Zero in Continuous Action Space","date":"2018-05-24","arxiv_id":"1805.09613","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a0c-alpha-zero-in-continuous-action-space#ran","syntology_url":"https://syntology.ai/paper/1805.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09613"}},"official":null}},{"url":"/paper/meta-gradient-reinforcement-learning","slug":"meta-gradient-reinforcement-learning","title":"Meta-Gradient Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09801","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meta-gradient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1805.09801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09801"}},"official":null}},{"url":"/paper/robust-distant-supervision-relation","slug":"robust-distant-supervision-relation","title":"Robust Distant Supervision Relation Extraction via Deep Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09927","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-distant-supervision-relation#ran","syntology_url":"https://syntology.ai/paper/1805.09927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09927"}},"official":null}}],"record_sha256":"b876980ea2aeb1aceb4216e0ee45f98b979c8552001ae7d06751ec3d3d9400fe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}