{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/ran/10","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":10,"pages_in_order":12,"rows_per_page":100,"rows":[901,1000],"of":1175,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning/papers/ran/1","prev":"/task/reinforcement-learning/papers/ran/9","next":"/task/reinforcement-learning/papers/ran/11","papers":[{"url":"/paper/deep-neuroevolution-of-recurrent-and-discrete","slug":"deep-neuroevolution-of-recurrent-and-discrete","title":"Deep Neuroevolution of Recurrent and Discrete World Models","date":"2019-04-28","arxiv_id":"1906.08857","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-neuroevolution-of-recurrent-and-discrete#ran","syntology_url":"https://syntology.ai/paper/1906.08857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.08857"}},"official":{"repos":["sebastianrisi/ga-world-models"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/faster-and-more-accurate-learning-with-meta","slug":"faster-and-more-accurate-learning-with-meta","title":"META-Learning State-based Eligibility Traces for More Sample-Efficient Policy Evaluation","date":"2019-04-25","arxiv_id":"1904.11439","repositories_listed":2,"syntology":{"n":14,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/faster-and-more-accurate-learning-with-meta#ran","syntology_url":"https://syntology.ai/paper/1904.11439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.11439"}},"official":{"repos":["PwnerHarry/META","PwnerHarry/MTA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-logic-reinforcement-learning","slug":"neural-logic-reinforcement-learning","title":"Neural Logic Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.10729","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neural-logic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1904.10729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10729"}},"official":{"repos":["ZhengyaoJiang/NLRL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/path-restore-learning-network-path-selection","slug":"path-restore-learning-network-path-selection","title":"Path-Restore: Learning Network Path Selection for Image Restoration","date":"2019-04-23","arxiv_id":"1904.10343","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/path-restore-learning-network-path-selection#ran","syntology_url":"https://syntology.ai/paper/1904.10343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10343"}},"official":null}},{"url":"/paper/graphnas-graph-neural-architecture-search","slug":"graphnas-graph-neural-architecture-search","title":"GraphNAS: Graph Neural Architecture Search with Reinforcement Learning","date":"2019-04-22","arxiv_id":"1904.09981","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/graphnas-graph-neural-architecture-search#ran","syntology_url":"https://syntology.ai/paper/1904.09981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09981"}},"official":{"repos":["GraphNAS/GraphNAS-simple"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rogue-gym-a-new-challenge-for-generalization","slug":"rogue-gym-a-new-challenge-for-generalization","title":"Rogue-Gym: A New Challenge for Generalization in Reinforcement Learning","date":"2019-04-17","arxiv_id":"1904.08129","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rogue-gym-a-new-challenge-for-generalization#ran","syntology_url":"https://syntology.ai/paper/1904.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08129"}},"official":{"repos":["kngwyu/rogue-gym","kngwyu/rogue-gym-agents-cog19"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hitchhikers-guide-to-statistical","slug":"a-hitchhikers-guide-to-statistical","title":"A Hitchhiker's Guide to Statistical Comparisons of Reinforcement Learning Algorithms","date":"2019-04-15","arxiv_id":"1904.06979","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hitchhikers-guide-to-statistical#ran","syntology_url":"https://syntology.ai/paper/1904.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06979"}},"official":{"repos":["flowersteam/rl_stats","ccolas/rl_stats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extrapolating-beyond-suboptimal","slug":"extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extrapolating-beyond-suboptimal#ran","syntology_url":"https://syntology.ai/paper/1904.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06387"}},"official":{"repos":["hiwonjoon/ICML2019-TREX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-navigate-unseen-environments-back","slug":"learning-to-navigate-unseen-environments-back","title":"Learning to Navigate Unseen Environments: Back Translation with Environmental Dropout","date":"2019-04-08","arxiv_id":"1904.04195","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-navigate-unseen-environments-back#ran","syntology_url":"https://syntology.ai/paper/1904.04195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.04195"}},"official":{"repos":["airsplay/R2R-EnvDrop"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-reinforcement-learning-via","slug":"interpretable-reinforcement-learning-via","title":"Optimization Methods for Interpretable Differentiable Decision Trees in Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09338","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interpretable-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/1903.09338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.09338"}},"official":null}},{"url":"/paper/efficient-off-policy-meta-reinforcement","slug":"efficient-off-policy-meta-reinforcement","title":"Efficient Off-Policy Meta-Reinforcement Learning via Probabilistic Context Variables","date":"2019-03-19","arxiv_id":"1903.08254","repositories_listed":7,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-off-policy-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1903.08254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.08254"}},"official":{"repos":["katerakelly/oyster"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/coacor-code-annotation-for-code-retrieval","slug":"coacor-code-annotation-for-code-retrieval","title":"CoaCor: Code Annotation for Code Retrieval with Reinforcement Learning","date":"2019-03-13","arxiv_id":"1904.00720","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/coacor-code-annotation-for-code-retrieval#ran","syntology_url":"https://syntology.ai/paper/1904.00720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.00720"}},"official":{"repos":["LittleYUYU/CoaCor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/universally-slimmable-networks-and-improved","slug":"universally-slimmable-networks-and-improved","title":"Universally Slimmable Networks and Improved Training Techniques","date":"2019-03-12","arxiv_id":"1903.05134","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/universally-slimmable-networks-and-improved#ran","syntology_url":"https://syntology.ai/paper/1903.05134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.05134"}},"official":{"repos":["JiahuiYu/slimmable_networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-reinforcement-learning-with-expert","slug":"hybrid-reinforcement-learning-with-expert","title":"Hybrid Reinforcement Learning with Expert State Sequences","date":"2019-03-11","arxiv_id":"1903.04110","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-reinforcement-learning-with-expert#ran","syntology_url":"https://syntology.ai/paper/1903.04110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04110"}},"official":{"repos":["XiaoxiaoGuo/tensor4rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stroke-based-artistic-rendering-agent-with","slug":"stroke-based-artistic-rendering-agent-with","title":"Learning to Paint With Model-based Deep Reinforcement Learning","date":"2019-03-11","arxiv_id":"1903.04411","repositories_listed":6,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/stroke-based-artistic-rendering-agent-with#ran","syntology_url":"https://syntology.ai/paper/1903.04411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04411"}},"official":{"repos":["hzwer/ICCV2019-LearningToPaint"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-2","slug":"multi-agent-deep-reinforcement-learning-for-2","title":"Multi-Agent Deep Reinforcement Learning for Large-scale Traffic Signal Control","date":"2019-03-11","arxiv_id":"1903.04527","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-deep-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/1903.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04527"}},"official":{"repos":["cts198859/deeprl_signal_control"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-power-system-emergency-control-using","slug":"adaptive-power-system-emergency-control-using","title":"Adaptive Power System Emergency Control using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03712","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-power-system-emergency-control-using#ran","syntology_url":"https://syntology.ai/paper/1903.03712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03712"}},"official":{"repos":["RLGC-Project/RLGC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minatar-an-atari-inspired-testbed-for-more","slug":"minatar-an-atari-inspired-testbed-for-more","title":"MinAtar: An Atari-Inspired Testbed for Thorough and Reproducible Reinforcement Learning Experiments","date":"2019-03-07","arxiv_id":"1903.03176","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/minatar-an-atari-inspired-testbed-for-more#ran","syntology_url":"https://syntology.ai/paper/1903.03176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03176"}},"official":{"repos":["kenjyoung/MinAtar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/using-natural-language-for-reward-shaping-in","slug":"using-natural-language-for-reward-shaping-in","title":"Using Natural Language for Reward Shaping in Reinforcement Learning","date":"2019-03-05","arxiv_id":"1903.02020","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/using-natural-language-for-reward-shaping-in#ran","syntology_url":"https://syntology.ai/paper/1903.02020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.02020"}},"official":null}},{"url":"/paper/the-streetlearn-environment-and-dataset","slug":"the-streetlearn-environment-and-dataset","title":"The StreetLearn Environment and Dataset","date":"2019-03-04","arxiv_id":"1903.01292","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-streetlearn-environment-and-dataset#ran","syntology_url":"https://syntology.ai/paper/1903.01292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.01292"}},"official":{"repos":["deepmind/streetlearn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-primitive-hierarchical-lifelong","slug":"model-primitive-hierarchical-lifelong","title":"Model Primitive Hierarchical Lifelong Reinforcement Learning","date":"2019-03-04","arxiv_id":"1903.01567","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-primitive-hierarchical-lifelong#ran","syntology_url":"https://syntology.ai/paper/1903.01567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.01567"}},"official":{"repos":["sisl/MPHRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-reinforcement-learning-for-atari","slug":"model-based-reinforcement-learning-for-atari","title":"Model-Based Reinforcement Learning for Atari","date":"2019-03-01","arxiv_id":"1903.00374","repositories_listed":2,"syntology":{"n":21,"n_ran":16,"n_constructed":8,"n_ran_checked":10,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"16 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-atari#ran","syntology_url":"https://syntology.ai/paper/1903.00374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.00374"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/function-space-particle-optimization-for","slug":"function-space-particle-optimization-for","title":"Function Space Particle Optimization for Bayesian Neural Networks","date":"2019-02-26","arxiv_id":"1902.09754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/function-space-particle-optimization-for#ran","syntology_url":"https://syntology.ai/paper/1902.09754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.09754"}},"official":{"repos":["thu-ml/fpovi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diagnosing-bottlenecks-in-deep-q-learning","slug":"diagnosing-bottlenecks-in-deep-q-learning","title":"Diagnosing Bottlenecks in Deep Q-learning Algorithms","date":"2019-02-26","arxiv_id":"1902.10250","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diagnosing-bottlenecks-in-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1902.10250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.10250"}},"official":null}},{"url":"/paper/rethinking-action-spaces-for-reinforcement","slug":"rethinking-action-spaces-for-reinforcement","title":"Rethinking Action Spaces for Reinforcement Learning in End-to-end Dialog Agents with Latent Variable Models","date":"2019-02-23","arxiv_id":"1902.08858","repositories_listed":3,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rethinking-action-spaces-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1902.08858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.08858"}},"official":{"repos":["snakeztc/NeuralDialog-LaRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-general-framework-for-structured-learning","slug":"a-general-framework-for-structured-learning","title":"A General Framework for Structured Learning of Mechanical Systems","date":"2019-02-22","arxiv_id":"1902.08705","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-general-framework-for-structured-learning#ran","syntology_url":"https://syntology.ai/paper/1902.08705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.08705"}},"official":{"repos":["sisl/mechamodlearn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/nutrition-and-health-data-for-cost-sensitive","slug":"nutrition-and-health-data-for-cost-sensitive","title":"Cost-Sensitive Diagnosis and Learning Leveraging Public Health Data","date":"2019-02-19","arxiv_id":"1902.07102","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nutrition-and-health-data-for-cost-sensitive#ran","syntology_url":"https://syntology.ai/paper/1902.07102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.07102"}},"official":{"repos":["mkachuee/Opportunistic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/190504100","slug":"190504100","title":"Deep Reinforcement Learning using Genetic Algorithm for Parameter Optimization","date":"2019-02-19","arxiv_id":"1905.04100","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/190504100#ran","syntology_url":"https://syntology.ai/paper/1905.04100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04100"}},"official":{"repos":["aralab-unr/ReinforcementLearningWithGA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/preferences-implicit-in-the-state-of-the","slug":"preferences-implicit-in-the-state-of-the","title":"Preferences Implicit in the State of the World","date":"2019-02-12","arxiv_id":"1902.04198","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/preferences-implicit-in-the-state-of-the#ran","syntology_url":"https://syntology.ai/paper/1902.04198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.04198"}},"official":{"repos":["HumanCompatibleAI/rlsp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-starcraft-multi-agent-challenge","slug":"the-starcraft-multi-agent-challenge","title":"The StarCraft Multi-Agent Challenge","date":"2019-02-11","arxiv_id":"1902.04043","repositories_listed":23,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/the-starcraft-multi-agent-challenge#ran","syntology_url":"https://syntology.ai/paper/1902.04043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.04043"}},"official":{"repos":["oxwhirl/pymarl","oxwhirl/smac"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/artificial-intelligence-for-prosthetics","slug":"artificial-intelligence-for-prosthetics","title":"Artificial Intelligence for Prosthetics - challenge solutions","date":"2019-02-07","arxiv_id":"1902.02441","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/artificial-intelligence-for-prosthetics#ran","syntology_url":"https://syntology.ai/paper/1902.02441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.02441"}},"official":{"repos":["iasawseen/MultiServerRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-consolidation-for-continual","slug":"policy-consolidation-for-continual","title":"Policy Consolidation for Continual Reinforcement Learning","date":"2019-02-01","arxiv_id":"1902.00255","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-consolidation-for-continual#ran","syntology_url":"https://syntology.ai/paper/1902.00255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00255"}},"official":null}},{"url":"/paper/learnable-embedding-space-for-efficient","slug":"learnable-embedding-space-for-efficient","title":"Learnable Embedding Space for Efficient Neural Architecture Compression","date":"2019-02-01","arxiv_id":"1902.00383","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learnable-embedding-space-for-efficient#ran","syntology_url":"https://syntology.ai/paper/1902.00383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00383"}},"official":{"repos":["Friedrich1006/ESNAC"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/tf-replicator-distributed-machine-learning","slug":"tf-replicator-distributed-machine-learning","title":"TF-Replicator: Distributed Machine Learning for Researchers","date":"2019-02-01","arxiv_id":"1902.00465","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tf-replicator-distributed-machine-learning#ran","syntology_url":"https://syntology.ai/paper/1902.00465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.00465"}},"official":{"repos":["tensorflow/community"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/go-explore-a-new-approach-for-hard","slug":"go-explore-a-new-approach-for-hard","title":"Go-Explore: a New Approach for Hard-Exploration Problems","date":"2019-01-30","arxiv_id":"1901.10995","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/go-explore-a-new-approach-for-hard#ran","syntology_url":"https://syntology.ai/paper/1901.10995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.10995"}},"official":null}},{"url":"/paper/making-deep-q-learning-methods-robust-to-time","slug":"making-deep-q-learning-methods-robust-to-time","title":"Making Deep Q-learning methods robust to time discretization","date":"2019-01-28","arxiv_id":"1901.09732","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-deep-q-learning-methods-robust-to-time#ran","syntology_url":"https://syntology.ai/paper/1901.09732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09732"}},"official":null}},{"url":"/paper/action-robust-reinforcement-learning-and","slug":"action-robust-reinforcement-learning-and","title":"Action Robust Reinforcement Learning and Applications in Continuous Control","date":"2019-01-26","arxiv_id":"1901.09184","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-robust-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/1901.09184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09184"}},"official":{"repos":["tesslerc/ActionRobustRL","icml2019-anonymous-author/Action-Robust-Reinforcement-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-generalized-recursive-reasoning","slug":"multi-agent-generalized-recursive-reasoning","title":"Modelling Bounded Rationality in Multi-Agent Interactions by Generalized Recursive Reasoning","date":"2019-01-26","arxiv_id":"1901.09216","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multi-agent-generalized-recursive-reasoning#ran","syntology_url":"https://syntology.ai/paper/1901.09216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09216"}},"official":{"repos":["ying-wen/gr2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-linguistic-phenomena-in-multi-agent","slug":"emergent-linguistic-phenomena-in-multi-agent","title":"Emergent Linguistic Phenomena in Multi-Agent Communication Games","date":"2019-01-25","arxiv_id":"1901.08706","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/emergent-linguistic-phenomena-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1901.08706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08706"}},"official":{"repos":["lgraesser/MultimodalGame"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fairness-with-dynamics","slug":"fairness-with-dynamics","title":"Algorithms for Fairness in Sequential Decision Making","date":"2019-01-24","arxiv_id":"1901.08568","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fairness-with-dynamics#ran","syntology_url":"https://syntology.ai/paper/1901.08568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08568"}},"official":{"repos":["wmgithub/fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-agile-and-dynamic-motor-skills-for","slug":"learning-agile-and-dynamic-motor-skills-for","title":"Learning agile and dynamic motor skills for legged robots","date":"2019-01-24","arxiv_id":"1901.08652","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-agile-and-dynamic-motor-skills-for#ran","syntology_url":"https://syntology.ai/paper/1901.08652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08652"}},"official":{"repos":["junja94/anymal_science_robotics_supplementary"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-multi-agent-reinforcement-learning-in","slug":"the-multi-agent-reinforcement-learning-in","title":"The Multi-Agent Reinforcement Learning in MalmÖ (MARLÖ) Competition","date":"2019-01-23","arxiv_id":"1901.08129","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-multi-agent-reinforcement-learning-in#ran","syntology_url":"https://syntology.ai/paper/1901.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08129"}},"official":null}},{"url":"/paper/causal-reasoning-from-meta-reinforcement","slug":"causal-reasoning-from-meta-reinforcement","title":"Causal Reasoning from Meta-reinforcement Learning","date":"2019-01-23","arxiv_id":"1901.08162","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causal-reasoning-from-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.08162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08162"}},"official":null}},{"url":"/paper/fast-accurate-and-lightweight-super","slug":"fast-accurate-and-lightweight-super","title":"Fast, Accurate and Lightweight Super-Resolution with Neural Architecture Search","date":"2019-01-22","arxiv_id":"1901.07261","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fast-accurate-and-lightweight-super#ran","syntology_url":"https://syntology.ai/paper/1901.07261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.07261"}},"official":{"repos":["falsr/FALSR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/improving-coordination-in-multi-agent-deep","slug":"improving-coordination-in-multi-agent-deep","title":"Improving Coordination in Small-Scale Multi-Agent Deep Reinforcement Learning through Memory-driven Communication","date":"2019-01-12","arxiv_id":"1901.03887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-coordination-in-multi-agent-deep#ran","syntology_url":"https://syntology.ai/paper/1901.03887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.03887"}},"official":null}},{"url":"/paper/risk-aware-active-inverse-reinforcement","slug":"risk-aware-active-inverse-reinforcement","title":"Risk-Aware Active Inverse Reinforcement Learning","date":"2019-01-08","arxiv_id":"1901.02161","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/risk-aware-active-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1901.02161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.02161"}},"official":{"repos":["Pearl-UTexas/ActiveVaR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-imbalanced#ran","syntology_url":"https://syntology.ai/paper/1901.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01379"}},"official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/opportunistic-learning-budgeted-cost","slug":"opportunistic-learning-budgeted-cost","title":"Opportunistic Learning: Budgeted Cost-Sensitive Learning from Data Streams","date":"2019-01-02","arxiv_id":"1901.00243","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opportunistic-learning-budgeted-cost#ran","syntology_url":"https://syntology.ai/paper/1901.00243","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.00243"}},"official":{"repos":["mkachuee/Opportunistic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-design-rna","slug":"learning-to-design-rna","title":"Learning to Design RNA","date":"2018-12-31","arxiv_id":"1812.11951","repositories_listed":5,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-design-rna#ran","syntology_url":"https://syntology.ai/paper/1812.11951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.11951"}},"official":{"repos":["automl/learna"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dopamine-a-research-framework-for-deep","slug":"dopamine-a-research-framework-for-deep","title":"Dopamine: A Research Framework for Deep Reinforcement Learning","date":"2018-12-14","arxiv_id":"1812.06110","repositories_listed":13,"syntology":{"n":16,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/dopamine-a-research-framework-for-deep#ran","syntology_url":"https://syntology.ai/paper/1812.06110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.06110"}},"official":{"repos":["google/dopamine"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/an-empirical-model-of-large-batch-training","slug":"an-empirical-model-of-large-batch-training","title":"An Empirical Model of Large-Batch Training","date":"2018-12-14","arxiv_id":"1812.06162","repositories_listed":11,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-empirical-model-of-large-batch-training#ran","syntology_url":"https://syntology.ai/paper/1812.06162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.06162"}},"official":null}},{"url":"/paper/soft-actor-critic-algorithms-and-applications","slug":"soft-actor-critic-algorithms-and-applications","title":"Soft Actor-Critic Algorithms and Applications","date":"2018-12-13","arxiv_id":"1812.05905","repositories_listed":52,"syntology":{"n":40,"n_ran":38,"n_constructed":0,"n_ran_checked":35,"n_instrument":3,"n_unverified":2,"n_honours":4,"n_violates":1,"n_no_contract":30,"n_pointer_only":5,"phrase":"38 ran (of which 0 constructed an object rather than computing a result; 35 with no instrument failure: 4 honoured, 1 violated, 30 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soft-actor-critic-algorithms-and-applications#ran","syntology_url":"https://syntology.ai/paper/1812.05905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.05905"}},"official":{"repos":["rail-berkeley/softlearning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/off-policy-deep-reinforcement-learning","slug":"off-policy-deep-reinforcement-learning","title":"Off-Policy Deep Reinforcement Learning without Exploration","date":"2018-12-07","arxiv_id":"1812.02900","repositories_listed":10,"syntology":{"n":14,"n_ran":14,"n_constructed":12,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":13,"n_pointer_only":9,"phrase":"14 ran (of which 12 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-policy-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1812.02900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.02900"}},"official":{"repos":["sfujim/BCQ"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/quantifying-generalization-in-reinforcement","slug":"quantifying-generalization-in-reinforcement","title":"Quantifying Generalization in Reinforcement Learning","date":"2018-12-06","arxiv_id":"1812.02341","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-generalization-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1812.02341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.02341"}},"official":{"repos":["openai/coinrun"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-compose-dynamic-tree-structures","slug":"learning-to-compose-dynamic-tree-structures","title":"Learning to Compose Dynamic Tree Structures for Visual Contexts","date":"2018-12-05","arxiv_id":"1812.01880","repositories_listed":6,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-to-compose-dynamic-tree-structures#ran","syntology_url":"https://syntology.ai/paper/1812.01880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.01880"}},"official":null}},{"url":"/paper/learning-to-learn-how-to-learn-self-adaptive","slug":"learning-to-learn-how-to-learn-self-adaptive","title":"Learning to Learn How to Learn: Self-Adaptive Visual Navigation Using Meta-Learning","date":"2018-12-03","arxiv_id":"1812.00971","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-learn-how-to-learn-self-adaptive#ran","syntology_url":"https://syntology.ai/paper/1812.00971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.00971"}},"official":{"repos":["allenai/savn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transferring-knowledge-across-learning","slug":"transferring-knowledge-across-learning","title":"Transferring Knowledge across Learning Processes","date":"2018-12-03","arxiv_id":"1812.01054","repositories_listed":4,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/transferring-knowledge-across-learning#ran","syntology_url":"https://syntology.ai/paper/1812.01054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.01054"}},"official":{"repos":["amzn/xfer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deep-multi-agent-reinforcement-learning-with","slug":"deep-multi-agent-reinforcement-learning-with","title":"Deep Multi-Agent Reinforcement Learning with Relevance Graphs","date":"2018-11-30","arxiv_id":"1811.12557","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/deep-multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1811.12557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.12557"}},"official":{"repos":["tegg89/magnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-impact-of-entropy-on-policy","slug":"understanding-the-impact-of-entropy-on-policy","title":"Understanding the impact of entropy on policy optimization","date":"2018-11-27","arxiv_id":"1811.11214","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/understanding-the-impact-of-entropy-on-policy#ran","syntology_url":"https://syntology.ai/paper/1811.11214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.11214"}},"official":{"repos":["zafarali/emdp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-uplift-modeling","slug":"reinforcement-learning-for-uplift-modeling","title":"Reinforcement Learning for Uplift Modeling","date":"2018-11-26","arxiv_id":"1811.10158","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-uplift-modeling#ran","syntology_url":"https://syntology.ai/paper/1811.10158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.10158"}},"official":null}},{"url":"/paper/improving-automatic-source-code-summarization","slug":"improving-automatic-source-code-summarization","title":"Improving Automatic Source Code Summarization via Deep Reinforcement Learning","date":"2018-11-17","arxiv_id":"1811.07234","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-automatic-source-code-summarization#ran","syntology_url":"https://syntology.ai/paper/1811.07234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.07234"}},"official":null}},{"url":"/paper/grasp2vec-learning-object-representations","slug":"grasp2vec-learning-object-representations","title":"Grasp2Vec: Learning Object Representations from Self-Supervised Grasping","date":"2018-11-16","arxiv_id":"1811.06964","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grasp2vec-learning-object-representations#ran","syntology_url":"https://syntology.ai/paper/1811.06964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.06964"}},"official":null}},{"url":"/paper/quota-the-quantile-option-architecture-for","slug":"quota-the-quantile-option-architecture-for","title":"QUOTA: The Quantile Option Architecture for Reinforcement Learning","date":"2018-11-05","arxiv_id":"1811.02073","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/quota-the-quantile-option-architecture-for#ran","syntology_url":"https://syntology.ai/paper/1811.02073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.02073"}},"official":{"repos":["ShangtongZhang/DeepRL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/exploration-by-random-network-distillation","slug":"exploration-by-random-network-distillation","title":"Exploration by Random Network Distillation","date":"2018-10-30","arxiv_id":"1810.12894","repositories_listed":22,"syntology":{"n":43,"n_ran":31,"n_constructed":12,"n_ran_checked":24,"n_instrument":7,"n_unverified":12,"n_honours":2,"n_violates":1,"n_no_contract":21,"n_pointer_only":16,"phrase":"31 ran (of which 12 constructed an object rather than computing a result; 24 with no instrument failure: 2 honoured, 1 violated, 21 with no contract checked; 7 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/exploration-by-random-network-distillation#ran","syntology_url":"https://syntology.ai/paper/1810.12894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12894"}},"official":{"repos":["openai/random-network-distillation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-to-learn-without-forgetting-by","slug":"learning-to-learn-without-forgetting-by","title":"Learning to Learn without Forgetting by Maximizing Transfer and Minimizing Interference","date":"2018-10-29","arxiv_id":"1810.11910","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-learn-without-forgetting-by#ran","syntology_url":"https://syntology.ai/paper/1810.11910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11910"}},"official":{"repos":["mattriemer/mer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-inference-with-tail-adaptive-f","slug":"variational-inference-with-tail-adaptive-f","title":"Variational Inference with Tail-adaptive f-Divergence","date":"2018-10-29","arxiv_id":"1810.11943","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/variational-inference-with-tail-adaptive-f#ran","syntology_url":"https://syntology.ai/paper/1810.11943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11943"}},"official":{"repos":["dilinwang820/adaptive-f-divergence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-active-exploration","slug":"model-based-active-exploration","title":"Model-Based Active Exploration","date":"2018-10-29","arxiv_id":"1810.12162","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-active-exploration#ran","syntology_url":"https://syntology.ai/paper/1810.12162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12162"}},"official":{"repos":["nnaisense/max"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/assessing-generalization-in-deep","slug":"assessing-generalization-in-deep","title":"Assessing Generalization in Deep Reinforcement Learning","date":"2018-10-29","arxiv_id":"1810.12282","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assessing-generalization-in-deep#ran","syntology_url":"https://syntology.ai/paper/1810.12282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.12282"}},"official":{"repos":["sunblaze-ucb/rl-generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-common-knowledge-reinforcement","slug":"multi-agent-common-knowledge-reinforcement","title":"Multi-Agent Common Knowledge Reinforcement Learning","date":"2018-10-27","arxiv_id":"1810.11702","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-common-knowledge-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1810.11702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11702"}},"official":{"repos":["schroederdewitt/mackrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-modular-control-for-embodied-question","slug":"neural-modular-control-for-embodied-question","title":"Neural Modular Control for Embodied Question Answering","date":"2018-10-26","arxiv_id":"1810.11181","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/neural-modular-control-for-embodied-question#ran","syntology_url":"https://syntology.ai/paper/1810.11181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.11181"}},"official":null}},{"url":"/paper/making-sense-of-vision-and-touch-self","slug":"making-sense-of-vision-and-touch-self","title":"Making Sense of Vision and Touch: Self-Supervised Learning of Multimodal Representations for Contact-Rich Tasks","date":"2018-10-24","arxiv_id":"1810.10191","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/making-sense-of-vision-and-touch-self#ran","syntology_url":"https://syntology.ai/paper/1810.10191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.10191"}},"official":{"repos":["stanford-iprl-lab/multimodal_representation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-convolutional-reinforcement-learning","slug":"graph-convolutional-reinforcement-learning","title":"Graph Convolutional Reinforcement Learning","date":"2018-10-22","arxiv_id":"1810.09202","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-convolutional-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.09202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09202"}},"official":{"repos":["PKU-AI-Edge/DGN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fast-deep-reinforcement-learning-using-online","slug":"fast-deep-reinforcement-learning-using-online","title":"Fast deep reinforcement learning using online adjustments from the past","date":"2018-10-18","arxiv_id":"1810.08163","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-deep-reinforcement-learning-using-online#ran","syntology_url":"https://syntology.ai/paper/1810.08163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.08163"}},"official":null}},{"url":"/paper/promp-proximal-meta-policy-search","slug":"promp-proximal-meta-policy-search","title":"ProMP: Proximal Meta-Policy Search","date":"2018-10-16","arxiv_id":"1810.06784","repositories_listed":6,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/promp-proximal-meta-policy-search#ran","syntology_url":"https://syntology.ai/paper/1810.06784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06784"}},"official":{"repos":["jonasrothfuss/promp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/successor-uncertainties-exploration-and","slug":"successor-uncertainties-exploration-and","title":"Successor Uncertainties: Exploration and Uncertainty in Temporal Difference Learning","date":"2018-10-15","arxiv_id":"1810.06530","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/successor-uncertainties-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1810.06530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06530"}},"official":null}},{"url":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parametrized-deep-q-networks-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06394"}},"official":null}},{"url":"/paper/fast-context-adaptation-via-meta-learning","slug":"fast-context-adaptation-via-meta-learning","title":"Fast Context Adaptation via Meta-Learning","date":"2018-10-08","arxiv_id":"1810.03642","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-context-adaptation-via-meta-learning#ran","syntology_url":"https://syntology.ai/paper/1810.03642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.03642"}},"official":{"repos":["lmzintgraf/cavia"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/where-did-my-optimum-go-an-empirical-analysis","slug":"where-did-my-optimum-go-an-empirical-analysis","title":"Where Did My Optimum Go?: An Empirical Analysis of Gradient Descent Optimization in Policy Gradient Methods","date":"2018-10-05","arxiv_id":"1810.02525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/where-did-my-optimum-go-an-empirical-analysis#ran","syntology_url":"https://syntology.ai/paper/1810.02525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02525"}},"official":{"repos":["facebookresearch/WhereDidMyOptimumGo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/actor-attention-critic-for-multi-agent","slug":"actor-attention-critic-for-multi-agent","title":"Actor-Attention-Critic for Multi-Agent Reinforcement Learning","date":"2018-10-05","arxiv_id":"1810.02912","repositories_listed":3,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-attention-critic-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1810.02912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02912"}},"official":{"repos":["shariqiqbal2810/MAAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/episodic-curiosity-through-reachability","slug":"episodic-curiosity-through-reachability","title":"Episodic Curiosity through Reachability","date":"2018-10-04","arxiv_id":"1810.02274","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/episodic-curiosity-through-reachability#ran","syntology_url":"https://syntology.ai/paper/1810.02274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02274"}},"official":{"repos":["google-research/episodic-curiosity"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-scheduling-algorithms-for-data","slug":"learning-scheduling-algorithms-for-data","title":"Learning Scheduling Algorithms for Data Processing Clusters","date":"2018-10-03","arxiv_id":"1810.01963","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-scheduling-algorithms-for-data#ran","syntology_url":"https://syntology.ai/paper/1810.01963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01963"}},"official":{"repos":["hongzimao/decima-sim"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-with-perturbed-rewards","slug":"reinforcement-learning-with-perturbed-rewards","title":"Reinforcement Learning with Perturbed Rewards","date":"2018-10-02","arxiv_id":"1810.01032","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-perturbed-rewards#ran","syntology_url":"https://syntology.ai/paper/1810.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01032"}},"official":{"repos":["wangjksjtu/rl-perturbed-reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emi-exploration-with-mutual-information","slug":"emi-exploration-with-mutual-information","title":"EMI: Exploration with Mutual Information","date":"2018-10-02","arxiv_id":"1810.01176","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emi-exploration-with-mutual-information#ran","syntology_url":"https://syntology.ai/paper/1810.01176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01176"}},"official":{"repos":["snu-mllab/EMI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-discriminator-bottleneck","slug":"variational-discriminator-bottleneck","title":"Variational Discriminator Bottleneck: Improving Imitation Learning, Inverse RL, and GANs by Constraining Information Flow","date":"2018-10-01","arxiv_id":"1810.00821","repositories_listed":5,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/variational-discriminator-bottleneck#ran","syntology_url":"https://syntology.ai/paper/1810.00821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.00821"}},"official":null}},{"url":"/paper/solving-statistical-mechanics-using","slug":"solving-statistical-mechanics-using","title":"Solving Statistical Mechanics Using Variational Autoregressive Networks","date":"2018-09-27","arxiv_id":"1809.10606","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/solving-statistical-mechanics-using#ran","syntology_url":"https://syntology.ai/paper/1809.10606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.10606"}},"official":{"repos":["wangleiphy/VAN.jl","wdphy16/stat-mech-van"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crowd-robot-interaction-crowd-aware-robot","slug":"crowd-robot-interaction-crowd-aware-robot","title":"Crowd-Robot Interaction: Crowd-aware Robot Navigation with Attention-based Deep Reinforcement Learning","date":"2018-09-24","arxiv_id":"1809.08835","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crowd-robot-interaction-crowd-aware-robot#ran","syntology_url":"https://syntology.ai/paper/1809.08835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.08835"}},"official":{"repos":["vita-epfl/CrowdNav","vita-epfl/DyNav"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/benchmarking-reinforcement-learning","slug":"benchmarking-reinforcement-learning","title":"Benchmarking Reinforcement Learning Algorithms on Real-World Robots","date":"2018-09-20","arxiv_id":"1809.07731","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1809.07731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.07731"}},"official":{"repos":["kindredresearch/SenseAct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-optimization-via-importance-sampling","slug":"policy-optimization-via-importance-sampling","title":"Policy Optimization via Importance Sampling","date":"2018-09-17","arxiv_id":"1809.06098","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-optimization-via-importance-sampling#ran","syntology_url":"https://syntology.ai/paper/1809.06098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06098"}},"official":{"repos":["T3p/pois"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/generalizing-across-multi-objective-reward","slug":"generalizing-across-multi-objective-reward","title":"Generalizing Across Multi-Objective Reward Functions in Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06364","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generalizing-across-multi-objective-reward#ran","syntology_url":"https://syntology.ai/paper/1809.06364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06364"}},"official":null}},{"url":"/paper/towards-better-interpretability-in-deep-q","slug":"towards-better-interpretability-in-deep-q","title":"Towards Better Interpretability in Deep Q-Networks","date":"2018-09-15","arxiv_id":"1809.05630","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-better-interpretability-in-deep-q#ran","syntology_url":"https://syntology.ai/paper/1809.05630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05630"}},"official":null}},{"url":"/paper/cm3-cooperative-multi-goal-multi-stage-multi","slug":"cm3-cooperative-multi-goal-multi-stage-multi","title":"CM3: Cooperative Multi-goal Multi-stage Multi-agent Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.05188","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cm3-cooperative-multi-goal-multi-stage-multi#ran","syntology_url":"https://syntology.ai/paper/1809.05188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05188"}},"official":{"repos":["011235813/cm3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with","slug":"multi-task-deep-reinforcement-learning-with","title":"Multi-task Deep Reinforcement Learning with PopArt","date":"2018-09-12","arxiv_id":"1809.04474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1809.04474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.04474"}},"official":null}},{"url":"/paper/improving-optimization-bounds-using-machine","slug":"improving-optimization-bounds-using-machine","title":"Improving Optimization Bounds using Machine Learning: Decision Diagrams meet Deep Reinforcement Learning","date":"2018-09-10","arxiv_id":"1809.03359","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-optimization-bounds-using-machine#ran","syntology_url":"https://syntology.ai/paper/1809.03359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.03359"}},"official":{"repos":["qcappart/learning-DD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/discriminator-actor-critic-addressing-sample","slug":"discriminator-actor-critic-addressing-sample","title":"Discriminator-Actor-Critic: Addressing Sample Inefficiency and Reward Bias in Adversarial Imitation Learning","date":"2018-09-09","arxiv_id":"1809.02925","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":4,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 4 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discriminator-actor-critic-addressing-sample#ran","syntology_url":"https://syntology.ai/paper/1809.02925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.02925"}},"official":null}},{"url":"/paper/unity-a-general-platform-for-intelligent","slug":"unity-a-general-platform-for-intelligent","title":"Unity: A General Platform for Intelligent Agents","date":"2018-09-07","arxiv_id":"1809.02627","repositories_listed":55,"syntology":{"n":44,"n_ran":29,"n_constructed":0,"n_ran_checked":25,"n_instrument":4,"n_unverified":15,"n_honours":0,"n_violates":0,"n_no_contract":25,"n_pointer_only":1,"phrase":"29 ran (of which 0 constructed an object rather than computing a result; 25 with no instrument failure: 0 honoured, 0 violated, 25 with no contract checked; 4 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/unity-a-general-platform-for-intelligent#ran","syntology_url":"https://syntology.ai/paper/1809.02627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.02627"}},"official":{"repos":["Unity-Technologies/ml-agents"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-hop-knowledge-graph-reasoning-with","slug":"multi-hop-knowledge-graph-reasoning-with","title":"Multi-Hop Knowledge Graph Reasoning with Reward Shaping","date":"2018-08-31","arxiv_id":"1808.10568","repositories_listed":3,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-hop-knowledge-graph-reasoning-with#ran","syntology_url":"https://syntology.ai/paper/1808.10568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.10568"}},"official":{"repos":["salesforce/MultiHopKG"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/decoupling-strategy-and-generation-in","slug":"decoupling-strategy-and-generation-in","title":"Decoupling Strategy and Generation in Negotiation Dialogues","date":"2018-08-29","arxiv_id":"1808.09637","repositories_listed":3,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/decoupling-strategy-and-generation-in#ran","syntology_url":"https://syntology.ai/paper/1808.09637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09637"}},"official":{"repos":["worksheets.codalab.org/worksheets/0x453913e76b65495d8b9730d41c7e0a0c"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/april-interactively-learning-to-summarise-by","slug":"april-interactively-learning-to-summarise-by","title":"APRIL: Interactively Learning to Summarise by Combining Active Preference Learning and Reinforcement Learning","date":"2018-08-29","arxiv_id":"1808.09658","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/april-interactively-learning-to-summarise-by#ran","syntology_url":"https://syntology.ai/paper/1808.09658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09658"}},"official":{"repos":["UKPLab/emnlp2018-april"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-study-of-reinforcement-learning-for-neural","slug":"a-study-of-reinforcement-learning-for-neural","title":"A Study of Reinforcement Learning for Neural Machine Translation","date":"2018-08-27","arxiv_id":"1808.08866","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-study-of-reinforcement-learning-for-neural#ran","syntology_url":"https://syntology.ai/paper/1808.08866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.08866"}},"official":{"repos":["apeterswu/RL4NMT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"db99c6ddee468b47cd4ceda8305337a23549988dcd2fa5749a45fb9a09ee27bf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}