{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/ran/8","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":8,"pages_in_order":12,"rows_per_page":100,"rows":[701,800],"of":1175,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning/papers/ran/1","prev":"/task/reinforcement-learning/papers/ran/7","next":"/task/reinforcement-learning/papers/ran/9","papers":[{"url":"/paper/guided-dialog-policy-learning-without","slug":"guided-dialog-policy-learning-without","title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop","date":"2020-04-07","arxiv_id":"2004.03267","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-dialog-policy-learning-without#ran","syntology_url":"https://syntology.ai/paper/2004.03267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.03267"}},"official":{"repos":["cszmli/dp-without-adv"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-2","slug":"multi-agent-reinforcement-learning-for-2","title":"Multi-agent Reinforcement Learning for Networked System Control","date":"2020-04-03","arxiv_id":"2004.01339","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":14,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":16,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/2004.01339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.01339"}},"official":{"repos":["cts198859/deeprl_network"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/action-space-shaping-in-deep-reinforcement","slug":"action-space-shaping-in-deep-reinforcement","title":"Action Space Shaping in Deep Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.00980","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-space-shaping-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.00980","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00980"}},"official":{"repos":["Miffyli/rl-action-space-shaping"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-sparse-rewarded-tasks-from-sub","slug":"learning-sparse-rewarded-tasks-from-sub","title":"Learning Sparse Rewarded Tasks from Sub-Optimal Demonstrations","date":"2020-04-01","arxiv_id":"2004.00530","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-rewarded-tasks-from-sub#ran","syntology_url":"https://syntology.ai/paper/2004.00530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00530"}},"official":null}},{"url":"/paper/ultrasound-guided-robotic-navigation-with","slug":"ultrasound-guided-robotic-navigation-with","title":"Ultrasound-Guided Robotic Navigation with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13321","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ultrasound-guided-robotic-navigation-with#ran","syntology_url":"https://syntology.ai/paper/2003.13321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13321"}},"official":{"repos":["hhase/spinal-navigation-rl","hhase/sacrum_data-set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/agent57-outperforming-the-atari-human","slug":"agent57-outperforming-the-atari-human","title":"Agent57: Outperforming the Atari Human Benchmark","date":"2020-03-30","arxiv_id":"2003.13350","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agent57-outperforming-the-atari-human#ran","syntology_url":"https://syntology.ai/paper/2003.13350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13350"}},"official":null}},{"url":"/paper/dhp-differentiable-meta-pruning-via","slug":"dhp-differentiable-meta-pruning-via","title":"DHP: Differentiable Meta Pruning via HyperNetworks","date":"2020-03-30","arxiv_id":"2003.13683","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dhp-differentiable-meta-pruning-via#ran","syntology_url":"https://syntology.ai/paper/2003.13683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13683"}},"official":{"repos":["ofsoundof/dhp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-ensemble-learning-with","slug":"sample-efficient-ensemble-learning-with","title":"Sample Efficient Ensemble Learning with Catalyst.RL","date":"2020-03-29","arxiv_id":"2003.14210","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sample-efficient-ensemble-learning-with#ran","syntology_url":"https://syntology.ai/paper/2003.14210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.14210"}},"official":{"repos":["Scitator/run-skeleton-run-in-3d"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/policy-teaching-via-environment-poisoning","slug":"policy-teaching-via-environment-poisoning","title":"Policy Teaching via Environment Poisoning: Training-time Adversarial Attacks against Reinforcement Learning","date":"2020-03-28","arxiv_id":"2003.12909","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-teaching-via-environment-poisoning#ran","syntology_url":"https://syntology.ai/paper/2003.12909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12909"}},"official":{"repos":["adishs/icml2020_rl-policy-teaching_code"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-3d-shapes-by-reinforcement-learning","slug":"modeling-3d-shapes-by-reinforcement-learning","title":"Modeling 3D Shapes by Reinforcement Learning","date":"2020-03-27","arxiv_id":"2003.12397","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-3d-shapes-by-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.12397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12397"}},"official":null}},{"url":"/paper/fiber-a-platform-for-efficient-development","slug":"fiber-a-platform-for-efficient-development","title":"Fiber: A Platform for Efficient Development and Distributed Training for Reinforcement Learning and Population-Based Methods","date":"2020-03-25","arxiv_id":"2003.11164","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fiber-a-platform-for-efficient-development#ran","syntology_url":"https://syntology.ai/paper/2003.11164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11164"}},"official":null}},{"url":"/paper/pads-policy-adapted-sampling-for-visual","slug":"pads-policy-adapted-sampling-for-visual","title":"PADS: Policy-Adapted Sampling for Visual Similarity Learning","date":"2020-03-24","arxiv_id":"2003.11113","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pads-policy-adapted-sampling-for-visual#ran","syntology_url":"https://syntology.ai/paper/2003.11113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11113"}},"official":{"repos":["Confusezius/CVPR2020_PADS"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-investigation-of-the-challenges","slug":"an-empirical-investigation-of-the-challenges","title":"An empirical investigation of the challenges of real-world reinforcement learning","date":"2020-03-24","arxiv_id":"2003.11881","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-empirical-investigation-of-the-challenges#ran","syntology_url":"https://syntology.ai/paper/2003.11881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11881"}},"official":{"repos":["google-research/realworldrl_suite"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-population-curriculum-for-1","slug":"evolutionary-population-curriculum-for-1","title":"Evolutionary Population Curriculum for Scaling Multi-Agent Reinforcement Learning","date":"2020-03-23","arxiv_id":"2003.10423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolutionary-population-curriculum-for-1#ran","syntology_url":"https://syntology.ai/paper/2003.10423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10423"}},"official":{"repos":["qian18long/epciclr2020"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhanced-poet-open-ended-reinforcement","slug":"enhanced-poet-open-ended-reinforcement","title":"Enhanced POET: Open-Ended Reinforcement Learning through Unbounded Invention of Learning Challenges and their Solutions","date":"2020-03-19","arxiv_id":"2003.08536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhanced-poet-open-ended-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.08536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08536"}},"official":{"repos":["uber-research/poet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monotonic-value-function-factorisation-for","slug":"monotonic-value-function-factorisation-for","title":"Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monotonic-value-function-factorisation-for#ran","syntology_url":"https://syntology.ai/paper/2003.08839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08839"}},"official":{"repos":["oxwhirl/pymarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neuroevolution-of-self-interpretable-agents","slug":"neuroevolution-of-self-interpretable-agents","title":"Neuroevolution of Self-Interpretable Agents","date":"2020-03-18","arxiv_id":"2003.08165","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neuroevolution-of-self-interpretable-agents#ran","syntology_url":"https://syntology.ai/paper/2003.08165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08165"}},"official":null}},{"url":"/paper/self-supervised-discovering-of-causal","slug":"self-supervised-discovering-of-causal","title":"Self-Supervised Discovering of Interpretable Features for Reinforcement Learning","date":"2020-03-16","arxiv_id":"2003.07069","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-discovering-of-causal#ran","syntology_url":"https://syntology.ai/paper/2003.07069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07069"}},"official":{"repos":["shiwj16/SSINet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discor-corrective-feedback-in-reinforcement","slug":"discor-corrective-feedback-in-reinforcement","title":"DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","date":"2020-03-16","arxiv_id":"2003.07305","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/discor-corrective-feedback-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.07305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07305"}},"official":null}},{"url":"/paper/target-driven-visual-navigation-exploiting","slug":"target-driven-visual-navigation-exploiting","title":"Learning hierarchical relationships for object-goal navigation","date":"2020-03-15","arxiv_id":"2003.06749","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/target-driven-visual-navigation-exploiting#ran","syntology_url":"https://syntology.ai/paper/2003.06749","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06749"}},"official":null}},{"url":"/paper/deep-multi-agent-reinforcement-learning-for","slug":"deep-multi-agent-reinforcement-learning-for","title":"FACMAC: Factored Multi-Agent Centralised Policy Gradients","date":"2020-03-14","arxiv_id":"2003.06709","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-multi-agent-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2003.06709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06709"}},"official":{"repos":["schroederdewitt/multiagent_mujoco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-deterministic-portfolio-optimization","slug":"deep-deterministic-portfolio-optimization","title":"Deep Deterministic Portfolio Optimization","date":"2020-03-13","arxiv_id":"2003.06497","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-deterministic-portfolio-optimization#ran","syntology_url":"https://syntology.ai/paper/2003.06497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06497"}},"official":{"repos":["CFMTech/Deep-RL-for-Portfolio-Optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-trojai-software-framework-an-opensource","slug":"the-trojai-software-framework-an-opensource","title":"The TrojAI Software Framework: An OpenSource tool for Embedding Trojans into Deep Learning Models","date":"2020-03-13","arxiv_id":"2003.07233","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-trojai-software-framework-an-opensource#ran","syntology_url":"https://syntology.ai/paper/2003.07233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07233"}},"official":null}},{"url":"/paper/sample-efficient-reinforcement-learning","slug":"sample-efficient-reinforcement-learning","title":"Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","date":"2020-03-12","arxiv_id":"2003.06066","repositories_listed":3,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sample-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.06066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06066"}},"official":null}},{"url":"/paper/meta-learning-curiosity-algorithms-1","slug":"meta-learning-curiosity-algorithms-1","title":"Meta-learning curiosity algorithms","date":"2020-03-11","arxiv_id":"2003.05325","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-learning-curiosity-algorithms-1#ran","syntology_url":"https://syntology.ai/paper/2003.05325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05325"}},"official":{"repos":["mfranzs/meta-learning-curiosity-algorithms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/online-meta-critic-learning-for-off-policy-1","slug":"online-meta-critic-learning-for-off-policy-1","title":"Online Meta-Critic Learning for Off-Policy Actor-Critic Methods","date":"2020-03-11","arxiv_id":"2003.05334","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/online-meta-critic-learning-for-off-policy-1#ran","syntology_url":"https://syntology.ai/paper/2003.05334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05334"}},"official":{"repos":["zwfightzw/Meta-Critic"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-minerl-competition-on-sample-efficient-1","slug":"the-minerl-competition-on-sample-efficient-1","title":"Retrospective Analysis of the 2019 MineRL Competition on Sample Efficient Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.05012","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-minerl-competition-on-sample-efficient-1#ran","syntology_url":"https://syntology.ai/paper/2003.05012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05012"}},"official":null}},{"url":"/paper/stable-policy-optimization-via-off-policy","slug":"stable-policy-optimization-via-off-policy","title":"Stable Policy Optimization via Off-Policy Divergence Regularization","date":"2020-03-09","arxiv_id":"2003.04108","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stable-policy-optimization-via-off-policy#ran","syntology_url":"https://syntology.ai/paper/2003.04108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04108"}},"official":{"repos":["facebookresearch/ppo-dice"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-discrete-state-abstractions-with","slug":"learning-discrete-state-abstractions-with","title":"Learning Discrete State Abstractions With Deep Variational Inference","date":"2020-03-09","arxiv_id":"2003.04300","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-discrete-state-abstractions-with#ran","syntology_url":"https://syntology.ai/paper/2003.04300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04300"}},"official":{"repos":["ondrejba/discrete_abstractions"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dada-differentiable-automatic-data","slug":"dada-differentiable-automatic-data","title":"DADA: Differentiable Automatic Data Augmentation","date":"2020-03-08","arxiv_id":"2003.03780","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dada-differentiable-automatic-data#ran","syntology_url":"https://syntology.ai/paper/2003.03780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.03780"}},"official":{"repos":["VDIGPKU/DADA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autophase-juggling-hls-phase-orderings-in","slug":"autophase-juggling-hls-phase-orderings-in","title":"AutoPhase: Juggling HLS Phase Orderings in Random Forests with Deep Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00671","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autophase-juggling-hls-phase-orderings-in#ran","syntology_url":"https://syntology.ai/paper/2003.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00671"}},"official":{"repos":["ucb-bar/autophase"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-co-learning-of-deep-and-spiking","slug":"reinforcement-co-learning-of-deep-and-spiking","title":"Reinforcement co-Learning of Deep and Spiking Neural Networks for Energy-Efficient Mapless Navigation with Neuromorphic Hardware","date":"2020-03-02","arxiv_id":"2003.01157","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-co-learning-of-deep-and-spiking#ran","syntology_url":"https://syntology.ai/paper/2003.01157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.01157"}},"official":{"repos":["combra-lab/spiking-ddpg-mapless-navigation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hybrid-stochastic-policy-gradient-algorithm","slug":"a-hybrid-stochastic-policy-gradient-algorithm","title":"A Hybrid Stochastic Policy Gradient Algorithm for Reinforcement Learning","date":"2020-03-01","arxiv_id":"2003.00430","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hybrid-stochastic-policy-gradient-algorithm#ran","syntology_url":"https://syntology.ai/paper/2003.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00430"}},"official":{"repos":["unc-optimization/ProxHSPGA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plannable-approximations-to-mdp-homomorphisms","slug":"plannable-approximations-to-mdp-homomorphisms","title":"Plannable Approximations to MDP Homomorphisms: Equivariance under Actions","date":"2020-02-27","arxiv_id":"2002.11963","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/plannable-approximations-to-mdp-homomorphisms#ran","syntology_url":"https://syntology.ai/paper/2002.11963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.11963"}},"official":{"repos":["ElisevanderPol/prae"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-exploration-even-with-a-1#ran","syntology_url":"https://syntology.ai/paper/2002.12174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.12174"}},"official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accessing-higher-level-representations-in","slug":"accessing-higher-level-representations-in","title":"Addressing Some Limitations of Transformers with Feedback Memory","date":"2020-02-21","arxiv_id":"2002.09402","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-higher-level-representations-in#ran","syntology_url":"https://syntology.ai/paper/2002.09402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.09402"}},"official":{"repos":["facebookresearch/transformer-sequential"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/how-to-avoid-being-eaten-by-a-grue","slug":"how-to-avoid-being-eaten-by-a-grue","title":"How To Avoid Being Eaten By a Grue: Exploration Strategies for Text-Adventure Agents","date":"2020-02-19","arxiv_id":"2002.08795","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-avoid-being-eaten-by-a-grue#ran","syntology_url":"https://syntology.ai/paper/2002.08795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08795"}},"official":{"repos":["rajammanabrolu/Q-BERT"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/generating-automatic-curricula-via-self","slug":"generating-automatic-curricula-via-self","title":"Generating Automatic Curricula via Self-Supervised Active Domain Randomization","date":"2020-02-18","arxiv_id":"2002.07911","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-automatic-curricula-via-self#ran","syntology_url":"https://syntology.ai/paper/2002.07911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.07911"}},"official":null}},{"url":"/paper/control-frequency-adaptation-via-action","slug":"control-frequency-adaptation-via-action","title":"Control Frequency Adaptation via Action Persistence in Batch Reinforcement Learning","date":"2020-02-17","arxiv_id":"2002.06836","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/control-frequency-adaptation-via-action#ran","syntology_url":"https://syntology.ai/paper/2002.06836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06836"}},"official":{"repos":["albertometelli/pfqi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/jelly-bean-world-a-testbed-for-never-ending-1","slug":"jelly-bean-world-a-testbed-for-never-ending-1","title":"Jelly Bean World: A Testbed for Never-Ending Learning","date":"2020-02-15","arxiv_id":"2002.06306","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/jelly-bean-world-a-testbed-for-never-ending-1#ran","syntology_url":"https://syntology.ai/paper/2002.06306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06306"}},"official":{"repos":["eaplatanios/jelly-bean-world"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pddlgym-gym-environments-from-pddl-problems","slug":"pddlgym-gym-environments-from-pddl-problems","title":"PDDLGym: Gym Environments from PDDL Problems","date":"2020-02-15","arxiv_id":"2002.06432","repositories_listed":2,"syntology":{"n":11,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/pddlgym-gym-environments-from-pddl-problems#ran","syntology_url":"https://syntology.ai/paper/2002.06432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06432"}},"official":{"repos":["tomsilver/pddlgym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/an-inductive-bias-for-distances-neural-nets-1","slug":"an-inductive-bias-for-distances-neural-nets-1","title":"An Inductive Bias for Distances: Neural Nets that Respect the Triangle Inequality","date":"2020-02-14","arxiv_id":"2002.05825","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-inductive-bias-for-distances-neural-nets-1#ran","syntology_url":"https://syntology.ai/paper/2002.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05825"}},"official":{"repos":["spitis/deepnorms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/never-give-up-learning-directed-exploration-1","slug":"never-give-up-learning-directed-exploration-1","title":"Never Give Up: Learning Directed Exploration Strategies","date":"2020-02-14","arxiv_id":"2002.06038","repositories_listed":6,"syntology":{"n":14,"n_ran":7,"n_constructed":3,"n_ran_checked":5,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/never-give-up-learning-directed-exploration-1#ran","syntology_url":"https://syntology.ai/paper/2002.06038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06038"}},"official":null}},{"url":"/paper/laprop-a-better-way-to-combine-momentum-with","slug":"laprop-a-better-way-to-combine-momentum-with","title":"LaProp: Separating Momentum and Adaptivity in Adam","date":"2020-02-12","arxiv_id":"2002.04839","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/laprop-a-better-way-to-combine-momentum-with#ran","syntology_url":"https://syntology.ai/paper/2002.04839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.04839"}},"official":{"repos":["Z-T-WANG/LaProp-Optimizer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fully-differentiable-procedural-content","slug":"fully-differentiable-procedural-content","title":"Learning to Generate Levels From Nothing","date":"2020-02-12","arxiv_id":"2002.05259","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fully-differentiable-procedural-content#ran","syntology_url":"https://syntology.ai/paper/2002.05259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05259"}},"official":{"repos":["pbontrager/GenerativePlayingNetworks"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-assttentive-associative-memory","slug":"self-assttentive-associative-memory","title":"Self-Attentive Associative Memory","date":"2020-02-10","arxiv_id":"2002.03519","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-assttentive-associative-memory#ran","syntology_url":"https://syntology.ai/paper/2002.03519","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03519"}},"official":{"repos":["thaihungle/SAM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discrete-action-on-policy-learning-with","slug":"discrete-action-on-policy-learning-with","title":"Discrete Action On-Policy Learning with Action-Value Critic","date":"2020-02-10","arxiv_id":"2002.03534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-action-on-policy-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.03534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03534"}},"official":{"repos":["yuguangyue/CARSM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-portfolio","slug":"reinforcement-learning-based-portfolio","title":"Reinforcement-Learning based Portfolio Management with Augmented Asset Movement Prediction States","date":"2020-02-09","arxiv_id":"2002.05780","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/reinforcement-learning-based-portfolio#ran","syntology_url":"https://syntology.ai/paper/2002.05780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05780"}},"official":null}},{"url":"/paper/multi-type-mean-field-reinforcement-learning","slug":"multi-type-mean-field-reinforcement-learning","title":"Multi Type Mean Field Reinforcement Learning","date":"2020-02-06","arxiv_id":"2002.02513","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-type-mean-field-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2002.02513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02513"}},"official":{"repos":["BorealisAI/mtmfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/one-shot-bayes-opt-with-probabilistic","slug":"one-shot-bayes-opt-with-probabilistic","title":"Provably Efficient Online Hyperparameter Optimization with Population-Based Bandits","date":"2020-02-06","arxiv_id":"2002.02518","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/one-shot-bayes-opt-with-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2002.02518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02518"}},"official":{"repos":["jparkerholder/PB2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-time","slug":"a-reinforcement-learning-framework-for-time","title":"Dynamic Causal Effects Evaluation in A/B Testing with a Reinforcement Learning Framework","date":"2020-02-05","arxiv_id":"2002.01711","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-reinforcement-learning-framework-for-time#ran","syntology_url":"https://syntology.ai/paper/2002.01711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01711"}},"official":{"repos":["callmespring/causalrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-diversity-in-population-based","slug":"effective-diversity-in-population-based","title":"Effective Diversity in Population Based Reinforcement Learning","date":"2020-02-03","arxiv_id":"2002.00632","repositories_listed":2,"syntology":{"n":12,"n_ran":10,"n_constructed":6,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":9,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/effective-diversity-in-population-based#ran","syntology_url":"https://syntology.ai/paper/2002.00632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00632"}},"official":{"repos":["jparkerholder/DvD_ES"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/integrating-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.00434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00434"}},"official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-the-systematic-reporting-of-the","slug":"towards-the-systematic-reporting-of-the","title":"Towards the Systematic Reporting of the Energy and Carbon Footprints of Machine Learning","date":"2020-01-31","arxiv_id":"2002.05651","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/towards-the-systematic-reporting-of-the#ran","syntology_url":"https://syntology.ai/paper/2002.05651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05651"}},"official":{"repos":["Breakend/ClimateChangeFromMachineLearningResearch","Breakend/experiment-impact-tracker"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pcgrl-procedural-content-generation-via","slug":"pcgrl-procedural-content-generation-via","title":"PCGRL: Procedural Content Generation via Reinforcement Learning","date":"2020-01-24","arxiv_id":"2001.09212","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pcgrl-procedural-content-generation-via#ran","syntology_url":"https://syntology.ai/paper/2001.09212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.09212"}},"official":{"repos":["amidos2006/gym-pcgrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/interpretable-end-to-end-urban-autonomous#ran","syntology_url":"https://syntology.ai/paper/2001.08726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08726"}},"official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gradient-surgery-for-multi-task-learning-1","slug":"gradient-surgery-for-multi-task-learning-1","title":"Gradient Surgery for Multi-Task Learning","date":"2020-01-19","arxiv_id":"2001.06782","repositories_listed":18,"syntology":{"n":10,"n_ran":7,"n_constructed":5,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gradient-surgery-for-multi-task-learning-1#ran","syntology_url":"https://syntology.ai/paper/2001.06782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06782"}},"official":{"repos":["tianheyu927/PCGrad"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/tree-structured-policy-based-progressive","slug":"tree-structured-policy-based-progressive","title":"Tree-Structured Policy based Progressive Reinforcement Learning for Temporally Language Grounding in Video","date":"2020-01-18","arxiv_id":"2001.06680","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tree-structured-policy-based-progressive#ran","syntology_url":"https://syntology.ai/paper/2001.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06680"}},"official":{"repos":["WuJie1010/TSP-PRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/popcorn-partially-observed-prediction","slug":"popcorn-partially-observed-prediction","title":"POPCORN: Partially Observed Prediction COnstrained ReiNforcement Learning","date":"2020-01-13","arxiv_id":"2001.04032","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/popcorn-partially-observed-prediction#ran","syntology_url":"https://syntology.ai/paper/2001.04032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04032"}},"official":{"repos":["dtak/POPCORN-POMDP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/statistical-inference-of-the-value-function","slug":"statistical-inference-of-the-value-function","title":"Statistical Inference of the Value Function for Reinforcement Learning in Infinite Horizon Settings","date":"2020-01-13","arxiv_id":"2001.04515","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/statistical-inference-of-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2001.04515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04515"}},"official":{"repos":["shengzhang37/SAVE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-forecasts-in-unknown-environments","slug":"trajectory-forecasts-in-unknown-environments","title":"Trajectory Forecasts in Unknown Environments Conditioned on Grid-Based Plans","date":"2020-01-03","arxiv_id":"2001.00735","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trajectory-forecasts-in-unknown-environments#ran","syntology_url":"https://syntology.ai/paper/2001.00735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00735"}},"official":{"repos":["nachiket92/P2T"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/making-sense-of-reinforcement-learning-and-1","slug":"making-sense-of-reinforcement-learning-and-1","title":"Making Sense of Reinforcement Learning and Probabilistic Inference","date":"2020-01-03","arxiv_id":"2001.00805","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-sense-of-reinforcement-learning-and-1#ran","syntology_url":"https://syntology.ai/paper/2001.00805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00805"}},"official":null}},{"url":"/paper/meta-reinforcement-learning-with-autonomous-1","slug":"meta-reinforcement-learning-with-autonomous-1","title":"Meta Reinforcement Learning with Autonomous Inference of Subtask Dependencies","date":"2020-01-01","arxiv_id":"2001.00248","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-reinforcement-learning-with-autonomous-1#ran","syntology_url":"https://syntology.ai/paper/2001.00248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00248"}},"official":{"repos":["srsohn/msgi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-conditioned-policies","slug":"reward-conditioned-policies","title":"Reward-Conditioned Policies","date":"2019-12-31","arxiv_id":"1912.13465","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-conditioned-policies#ran","syntology_url":"https://syntology.ai/paper/1912.13465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.13465"}},"official":null}},{"url":"/paper/pac-confidence-sets-for-deep-neural-networks-1","slug":"pac-confidence-sets-for-deep-neural-networks-1","title":"PAC Confidence Sets for Deep Neural Networks via Calibrated Prediction","date":"2019-12-31","arxiv_id":"2001.00106","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/pac-confidence-sets-for-deep-neural-networks-1#ran","syntology_url":"https://syntology.ai/paper/2001.00106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00106"}},"official":{"repos":["sangdon/PAC-confidence-set"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-practical-multi-object-manipulation","slug":"towards-practical-multi-object-manipulation","title":"Towards Practical Multi-Object Manipulation using Relational Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.11032","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-practical-multi-object-manipulation#ran","syntology_url":"https://syntology.ai/paper/1912.11032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11032"}},"official":null}},{"url":"/paper/discrete-and-continuous-action-representation","slug":"discrete-and-continuous-action-representation","title":"Discrete and Continuous Action Representation for Practical RL in Video Games","date":"2019-12-23","arxiv_id":"1912.11077","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-and-continuous-action-representation#ran","syntology_url":"https://syntology.ai/paper/1912.11077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11077"}},"official":null}},{"url":"/paper/explain-your-move-understanding-agent-actions-1","slug":"explain-your-move-understanding-agent-actions-1","title":"Explain Your Move: Understanding Agent Actions Using Specific and Relevant Feature Attribution","date":"2019-12-23","arxiv_id":"1912.12191","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explain-your-move-understanding-agent-actions-1#ran","syntology_url":"https://syntology.ai/paper/1912.12191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12191"}},"official":{"repos":["rl-interpretation/understandingRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/interestingness-elements-for-explainable","slug":"interestingness-elements-for-explainable","title":"Interestingness Elements for Explainable Reinforcement Learning: Understanding Agents' Capabilities and Limitations","date":"2019-12-19","arxiv_id":"1912.09007","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interestingness-elements-for-explainable#ran","syntology_url":"https://syntology.ai/paper/1912.09007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.09007"}},"official":{"repos":["SRI-AIC/InterestingnessXRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reinforcement-learning-for-1","slug":"distributional-reinforcement-learning-for-1","title":"Distributional Reinforcement Learning for Energy-Based Sequential Models","date":"2019-12-18","arxiv_id":"1912.08517","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/distributional-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/1912.08517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.08517"}},"official":{"repos":["parshakova/GAMS-for-Data-Efficient-Learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dota-2-with-large-scale-deep-reinforcement","slug":"dota-2-with-large-scale-deep-reinforcement","title":"Dota 2 with Large Scale Deep Reinforcement Learning","date":"2019-12-13","arxiv_id":"1912.06680","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dota-2-with-large-scale-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1912.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.06680"}},"official":null}},{"url":"/paper/measuring-the-reliability-of-reinforcement-1","slug":"measuring-the-reliability-of-reinforcement-1","title":"Measuring the Reliability of Reinforcement Learning Algorithms","date":"2019-12-10","arxiv_id":"1912.05663","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-the-reliability-of-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1912.05663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05663"}},"official":{"repos":["google-research/rl-reliability-metrics"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chainerrl-a-deep-reinforcement-learning","slug":"chainerrl-a-deep-reinforcement-learning","title":"ChainerRL: A Deep Reinforcement Learning Library","date":"2019-12-09","arxiv_id":"1912.03905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chainerrl-a-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1912.03905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03905"}},"official":{"repos":["chainer/chainerrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-object-detection-in-large-images","slug":"efficient-object-detection-in-large-images","title":"Efficient Object Detection in Large Images using Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.03966","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-object-detection-in-large-images#ran","syntology_url":"https://syntology.ai/paper/1912.03966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03966"}},"official":{"repos":["uzkent/EfficientObjectDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/covariance-matrix-adaptation-for-the-rapid","slug":"covariance-matrix-adaptation-for-the-rapid","title":"Covariance Matrix Adaptation for the Rapid Illumination of Behavior Space","date":"2019-12-05","arxiv_id":"1912.02400","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/covariance-matrix-adaptation-for-the-rapid#ran","syntology_url":"https://syntology.ai/paper/1912.02400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.02400"}},"official":{"repos":["tehqin/EvoStone","tehqin/QualDivBenchmark"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-human-objectives-by-evaluating","slug":"learning-human-objectives-by-evaluating","title":"Learning Human Objectives by Evaluating Hypothetical Behavior","date":"2019-12-05","arxiv_id":"1912.05652","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-human-objectives-by-evaluating#ran","syntology_url":"https://syntology.ai/paper/1912.05652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05652"}},"official":null}},{"url":"/paper/safelife-10-exploring-side-effects-in-complex","slug":"safelife-10-exploring-side-effects-in-complex","title":"SafeLife 1.0: Exploring Side Effects in Complex Environments","date":"2019-12-03","arxiv_id":"1912.01217","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safelife-10-exploring-side-effects-in-complex#ran","syntology_url":"https://syntology.ai/paper/1912.01217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01217"}},"official":{"repos":["PartnershipOnAI/safelife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-procedural-generation-to-benchmark","slug":"leveraging-procedural-generation-to-benchmark","title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01588","repositories_listed":6,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leveraging-procedural-generation-to-benchmark#ran","syntology_url":"https://syntology.ai/paper/1912.01588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01588"}},"official":{"repos":["openai/procgen","openai/train-procgen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dream-to-control-learning-behaviors-by-latent","slug":"dream-to-control-learning-behaviors-by-latent","title":"Dream to Control: Learning Behaviors by Latent Imagination","date":"2019-12-03","arxiv_id":"1912.01603","repositories_listed":21,"syntology":{"n":62,"n_ran":52,"n_constructed":33,"n_ran_checked":45,"n_instrument":7,"n_unverified":10,"n_honours":1,"n_violates":0,"n_no_contract":44,"n_pointer_only":9,"phrase":"52 ran (of which 33 constructed an object rather than computing a result; 45 with no instrument failure: 1 honoured, 0 violated, 44 with no contract checked; 7 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/dream-to-control-learning-behaviors-by-latent#ran","syntology_url":"https://syntology.ai/paper/1912.01603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01603"}},"official":{"repos":["danijar/dreamer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/playing-games-in-the-dark-an-approach-for","slug":"playing-games-in-the-dark-an-approach-for","title":"Playing Games in the Dark: An approach for cross-modality transfer in reinforcement learning","date":"2019-11-28","arxiv_id":"1911.12851","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-games-in-the-dark-an-approach-for#ran","syntology_url":"https://syntology.ai/paper/1911.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12851"}},"official":null}},{"url":"/paper/end-to-end-model-free-reinforcement-learning","slug":"end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","date":"2019-11-25","arxiv_id":"1911.10868","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-model-free-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.10868"}},"official":{"repos":["valeoai/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-with-goal-conditioned-policies-1","slug":"planning-with-goal-conditioned-policies-1","title":"Planning with Goal-Conditioned Policies","date":"2019-11-19","arxiv_id":"1911.08453","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planning-with-goal-conditioned-policies-1#ran","syntology_url":"https://syntology.ai/paper/1911.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08453"}},"official":null}},{"url":"/paper/combinatorial-optimization-by-graph-pointer","slug":"combinatorial-optimization-by-graph-pointer","title":"Combinatorial Optimization by Graph Pointer Networks and Hierarchical Reinforcement Learning","date":"2019-11-12","arxiv_id":"1911.04936","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/combinatorial-optimization-by-graph-pointer#ran","syntology_url":"https://syntology.ai/paper/1911.04936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04936"}},"official":{"repos":["qiang-ma/graph-pointer-network"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/drills-deep-reinforcement-learning-for-logic","slug":"drills-deep-reinforcement-learning-for-logic","title":"DRiLLS: Deep Reinforcement Learning for Logic Synthesis","date":"2019-11-11","arxiv_id":"1911.04021","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drills-deep-reinforcement-learning-for-logic#ran","syntology_url":"https://syntology.ai/paper/1911.04021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04021"}},"official":{"repos":["scale-lab/DRiLLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-connected-autonomous-driving","slug":"multi-agent-connected-autonomous-driving","title":"Multi-Agent Connected Autonomous Driving using Deep Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04175","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-connected-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/1911.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04175"}},"official":{"repos":["praveen-palanisamy/macad-gym"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/real-time-reinforcement-learning","slug":"real-time-reinforcement-learning","title":"Real-Time Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04448","repositories_listed":3,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.04448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04448"}},"official":{"repos":["rmst/rtrl","elementai/avenue"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-model-based-reinforcement-learning-with","slug":"a-model-based-reinforcement-learning-with","title":"Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-11-10","arxiv_id":"1911.03845","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-model-based-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1911.03845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.03845"}},"official":{"repos":["JianGuanTHU/IRecGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-divergence-minimization-perspective-on","slug":"a-divergence-minimization-perspective-on","title":"A Divergence Minimization Perspective on Imitation Learning Methods","date":"2019-11-06","arxiv_id":"1911.02256","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-divergence-minimization-perspective-on#ran","syntology_url":"https://syntology.ai/paper/1911.02256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02256"}},"official":{"repos":["KamyarGh/rl_swiss"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fully-parameterized-quantile-function-for","slug":"fully-parameterized-quantile-function-for","title":"Fully Parameterized Quantile Function for Distributional Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.02140","repositories_listed":6,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fully-parameterized-quantile-function-for#ran","syntology_url":"https://syntology.ai/paper/1911.02140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02140"}},"official":null}},{"url":"/paper/multimodal-model-agnostic-meta-learning-via","slug":"multimodal-model-agnostic-meta-learning-via","title":"Multimodal Model-Agnostic Meta-Learning via Task-Aware Modulation","date":"2019-10-30","arxiv_id":"1910.13616","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-model-agnostic-meta-learning-via#ran","syntology_url":"https://syntology.ai/paper/1910.13616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13616"}},"official":{"repos":["vuoristo/MMAML"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-to-predict-without-looking-ahead","slug":"learning-to-predict-without-looking-ahead","title":"Learning to Predict Without Looking Ahead: World Models Without Forward Prediction","date":"2019-10-29","arxiv_id":"1910.13038","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-predict-without-looking-ahead#ran","syntology_url":"https://syntology.ai/paper/1910.13038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13038"}},"official":null}},{"url":"/paper/191013439","slug":"191013439","title":"Learning to Manipulate Deformable Objects without Demonstrations","date":"2019-10-29","arxiv_id":"1910.13439","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/191013439#ran","syntology_url":"https://syntology.ai/paper/1910.13439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13439"}},"official":{"repos":["wilson1yan/rlpyt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/entity-abstraction-in-visual-model-based","slug":"entity-abstraction-in-visual-model-based","title":"Entity Abstraction in Visual Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/entity-abstraction-in-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/1910.12827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12827"}},"official":{"repos":["jcoreyes/OP3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zpd-teaching-strategies-for-deep","slug":"zpd-teaching-strategies-for-deep","title":"ZPD Teaching Strategies for Deep Reinforcement Learning from Demonstrations","date":"2019-10-26","arxiv_id":"1910.12154","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zpd-teaching-strategies-for-deep#ran","syntology_url":"https://syntology.ai/paper/1910.12154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12154"}},"official":null}},{"url":"/paper/bananas-bayesian-optimization-with-neural","slug":"bananas-bayesian-optimization-with-neural","title":"BANANAS: Bayesian Optimization with Neural Architectures for Neural Architecture Search","date":"2019-10-25","arxiv_id":"1910.11858","repositories_listed":3,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bananas-bayesian-optimization-with-neural#ran","syntology_url":"https://syntology.ai/paper/1910.11858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11858"}},"official":{"repos":["naszilla/bananas","naszilla/naszilla"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-world-a-benchmark-and-evaluation-for","slug":"meta-world-a-benchmark-and-evaluation-for","title":"Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.10897","repositories_listed":9,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meta-world-a-benchmark-and-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/1910.10897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10897"}},"official":{"repos":["rlworkgroup/metaworld"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/hrl4in-hierarchical-reinforcement-learning","slug":"hrl4in-hierarchical-reinforcement-learning","title":"HRL4IN: Hierarchical Reinforcement Learning for Interactive Navigation with Mobile Manipulators","date":"2019-10-24","arxiv_id":"1910.11432","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hrl4in-hierarchical-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.11432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11432"}},"official":null}},{"url":"/paper/robust-domain-randomization-for-reinforcement","slug":"robust-domain-randomization-for-reinforcement","title":"Robust Visual Domain Randomization for Reinforcement Learning","date":"2019-10-23","arxiv_id":"1910.10537","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-domain-randomization-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1910.10537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10537"}},"official":{"repos":["IndustAI/visual-domain-randomization","uncharted-technologies/robust-domain-randomization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"9226bd466702cd75fd79ad8ca4fa6843f9b29f74b0d56dd404d2a037ad9ab3f5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}