{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/ran/11","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":11,"pages_in_order":15,"rows_per_page":100,"rows":[1001,1100],"of":1416,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1/papers/ran/1","prev":"/task/reinforcement-learning-1/papers/ran/10","next":"/task/reinforcement-learning-1/papers/ran/12","papers":[{"url":"/paper/mdp-homomorphic-networks-group-symmetries-in","slug":"mdp-homomorphic-networks-group-symmetries-in","title":"MDP Homomorphic Networks: Group Symmetries in Reinforcement Learning","date":"2020-06-30","arxiv_id":"2006.16908","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-homomorphic-networks-group-symmetries-in#ran","syntology_url":"https://syntology.ai/paper/2006.16908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16908"}},"official":{"repos":["ElisevanderPol/mdp-homomorphic-networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/disk-learning-local-features-with-policy","slug":"disk-learning-local-features-with-policy","title":"DISK: Learning local features with policy gradient","date":"2020-06-24","arxiv_id":"2006.13566","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":4,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/disk-learning-local-features-with-policy#ran","syntology_url":"https://syntology.ai/paper/2006.13566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13566"}},"official":{"repos":["cvlab-epfl/disk"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-nethack-learning-environment","slug":"the-nethack-learning-environment","title":"The NetHack Learning Environment","date":"2020-06-24","arxiv_id":"2006.13760","repositories_listed":3,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/the-nethack-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2006.13760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13760"}},"official":{"repos":["facebookresearch/nle"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/off-dynamics-reinforcement-learning-training","slug":"off-dynamics-reinforcement-learning-training","title":"Off-Dynamics Reinforcement Learning: Training for Transfer with Domain Classifiers","date":"2020-06-24","arxiv_id":"2006.13916","repositories_listed":2,"syntology":{"n":23,"n_ran":21,"n_constructed":4,"n_ran_checked":15,"n_instrument":6,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":12,"n_pointer_only":7,"phrase":"21 ran (of which 4 constructed an object rather than computing a result; 15 with no instrument failure: 3 honoured, 0 violated, 12 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/off-dynamics-reinforcement-learning-training#ran","syntology_url":"https://syntology.ai/paper/2006.13916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13916"}},"official":null}},{"url":"/paper/automatic-data-augmentation-for","slug":"automatic-data-augmentation-for","title":"Automatic Data Augmentation for Generalization in Deep Reinforcement Learning","date":"2020-06-23","arxiv_id":"2006.12862","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automatic-data-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2006.12862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.12862"}},"official":{"repos":["rraileanu/auto-drac"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/experience-replay-with-likelihood-free","slug":"experience-replay-with-likelihood-free","title":"Experience Replay with Likelihood-free Importance Weights","date":"2020-06-23","arxiv_id":"2006.13169","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/experience-replay-with-likelihood-free#ran","syntology_url":"https://syntology.ai/paper/2006.13169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13169"}},"official":null}},{"url":"/paper/expert-supervised-reinforcement-learning-for","slug":"expert-supervised-reinforcement-learning-for","title":"Expert-Supervised Reinforcement Learning for Offline Policy Learning and Evaluation","date":"2020-06-23","arxiv_id":"2006.13189","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":2,"n_ran_checked":2,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":11,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/expert-supervised-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2006.13189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.13189"}},"official":{"repos":["asonabend/ESRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-with-amigo-adversarially-motivated","slug":"learning-with-amigo-adversarially-motivated","title":"Learning with AMIGo: Adversarially Motivated Intrinsic Goals","date":"2020-06-22","arxiv_id":"2006.12122","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-with-amigo-adversarially-motivated#ran","syntology_url":"https://syntology.ai/paper/2006.12122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.12122"}},"official":{"repos":["facebookresearch/adversarially-motivated-intrinsic-goals"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-factory-egocentric-3d-control-from","slug":"sample-factory-egocentric-3d-control-from","title":"Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning","date":"2020-06-21","arxiv_id":"2006.11751","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-factory-egocentric-3d-control-from#ran","syntology_url":"https://syntology.ai/paper/2006.11751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11751"}},"official":{"repos":["alex-petrenko/sample-factory"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-adjacency-constrained-subgoals-in","slug":"generating-adjacency-constrained-subgoals-in","title":"Generating Adjacency-Constrained Subgoals in Hierarchical Reinforcement Learning","date":"2020-06-20","arxiv_id":"2006.11485","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generating-adjacency-constrained-subgoals-in#ran","syntology_url":"https://syntology.ai/paper/2006.11485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11485"}},"official":{"repos":["trzhang0116/HRAC"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/nnc-neural-network-control-of-dynamical","slug":"nnc-neural-network-control-of-dynamical","title":"Neural Ordinary Differential Equation Control of Dynamics on Graphs","date":"2020-06-17","arxiv_id":"2006.09773","repositories_listed":1,"syntology":{"n":18,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nnc-neural-network-control-of-dynamical#ran","syntology_url":"https://syntology.ai/paper/2006.09773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09773"}},"official":{"repos":["asikist/nnc"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accelerating-online-reinforcement-learning","slug":"accelerating-online-reinforcement-learning","title":"AWAC: Accelerating Online Reinforcement Learning with Offline Datasets","date":"2020-06-16","arxiv_id":"2006.09359","repositories_listed":6,"syntology":{"n":13,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":8,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/accelerating-online-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.09359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09359"}},"official":{"repos":["vitchyr/rlkit"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/opponent-modelling-with-local-information","slug":"opponent-modelling-with-local-information","title":"Agent Modelling under Partial Observability for Deep Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.09447","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/opponent-modelling-with-local-information#ran","syntology_url":"https://syntology.ai/paper/2006.09447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09447"}},"official":{"repos":["uoe-agents/LIAM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learn-to-effectively-explore-in-context-based","slug":"learn-to-effectively-explore-in-context-based","title":"MetaCURE: Meta Reinforcement Learning with Empowerment-Driven Exploration","date":"2020-06-15","arxiv_id":"2006.08170","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learn-to-effectively-explore-in-context-based#ran","syntology_url":"https://syntology.ai/paper/2006.08170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08170"}},"official":{"repos":["NagisaZj/MetaCURE-Public"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pipeline-psro-a-scalable-approach-for-finding","slug":"pipeline-psro-a-scalable-approach-for-finding","title":"Pipeline PSRO: A Scalable Approach for Finding Approximate Nash Equilibria in Large Games","date":"2020-06-15","arxiv_id":"2006.08555","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pipeline-psro-a-scalable-approach-for-finding#ran","syntology_url":"https://syntology.ai/paper/2006.08555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08555"}},"official":{"repos":["JBLanier/pipeline-psro"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-distributionally-robust-policy","slug":"optimistic-distributionally-robust-policy","title":"Optimistic Distributionally Robust Policy Optimization","date":"2020-06-14","arxiv_id":"2006.07815","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimistic-distributionally-robust-policy#ran","syntology_url":"https://syntology.ai/paper/2006.07815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07815"}},"official":{"repos":["kadysongbb/dr-trpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/comparative-evaluation-of-multi-agent-deep","slug":"comparative-evaluation-of-multi-agent-deep","title":"Benchmarking Multi-Agent Deep Reinforcement Learning Algorithms in Cooperative Tasks","date":"2020-06-14","arxiv_id":"2006.07869","repositories_listed":9,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comparative-evaluation-of-multi-agent-deep#ran","syntology_url":"https://syntology.ai/paper/2006.07869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07869"}},"official":{"repos":["uoe-agents/epymarl","uoe-agents/robotic-warehouse","uoe-agents/lb-foraging"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/torsionnet-a-reinforcement-learning-approach","slug":"torsionnet-a-reinforcement-learning-approach","title":"TorsionNet: A Reinforcement Learning Approach to Sequential Conformer Search","date":"2020-06-12","arxiv_id":"2006.07078","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/torsionnet-a-reinforcement-learning-approach#ran","syntology_url":"https://syntology.ai/paper/2006.07078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07078"}},"official":{"repos":["tarungog/torsionnet_paper_version"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/distributed-reinforcement-learning-in-multi","slug":"distributed-reinforcement-learning-in-multi","title":"Multi-Agent Reinforcement Learning in Stochastic Networked Systems","date":"2020-06-11","arxiv_id":"2006.06555","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distributed-reinforcement-learning-in-multi#ran","syntology_url":"https://syntology.ai/paper/2006.06555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06555"}},"official":null}},{"url":"/paper/closed-loop-neural-symbolic-learning-via","slug":"closed-loop-neural-symbolic-learning-via","title":"Closed Loop Neural-Symbolic Learning via Integrating Neural Perception, Grammar Parsing, and Symbolic Reasoning","date":"2020-06-11","arxiv_id":"2006.06649","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/closed-loop-neural-symbolic-learning-via#ran","syntology_url":"https://syntology.ai/paper/2006.06649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06649"}},"official":{"repos":["liqing-ustc/NGS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/what-matters-in-on-policy-reinforcement","slug":"what-matters-in-on-policy-reinforcement","title":"What Matters In On-Policy Reinforcement Learning? A Large-Scale Empirical Study","date":"2020-06-10","arxiv_id":"2006.05990","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-matters-in-on-policy-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2006.05990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05990"}},"official":null}},{"url":"/paper/constrained-episodic-reinforcement-learning","slug":"constrained-episodic-reinforcement-learning","title":"Constrained episodic reinforcement learning in concave-convex and knapsack settings","date":"2020-06-09","arxiv_id":"2006.05051","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/constrained-episodic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.05051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.05051"}},"official":{"repos":["miryoosefi/ConRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-play-no-press-diplomacy-with-best","slug":"learning-to-play-no-press-diplomacy-with-best","title":"Learning to Play No-Press Diplomacy with Best Response Policy Iteration","date":"2020-06-08","arxiv_id":"2006.04635","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-play-no-press-diplomacy-with-best#ran","syntology_url":"https://syntology.ai/paper/2006.04635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04635"}},"official":{"repos":["deepmind/diplomacy"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-under-moral","slug":"reinforcement-learning-under-moral","title":"Reinforcement Learning Under Moral Uncertainty","date":"2020-06-08","arxiv_id":"2006.04734","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-under-moral#ran","syntology_url":"https://syntology.ai/paper/2006.04734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04734"}},"official":{"repos":["uber-research/normative-uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-q-learning-for-offline","slug":"conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","arxiv_id":"2006.04779","repositories_listed":18,"syntology":{"n":34,"n_ran":31,"n_constructed":5,"n_ran_checked":28,"n_instrument":3,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":26,"n_pointer_only":19,"phrase":"31 ran (of which 5 constructed an object rather than computing a result; 28 with no instrument failure: 2 honoured, 0 violated, 26 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/conservative-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2006.04779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04779"}},"official":{"repos":["aviralkumar2907/CQL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/reinforcement-learning-for-multi-product","slug":"reinforcement-learning-for-multi-product","title":"Reinforcement Learning for Multi-Product Multi-Node Inventory Management in Supply Chains","date":"2020-06-07","arxiv_id":"2006.04037","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-for-multi-product#ran","syntology_url":"https://syntology.ai/paper/2006.04037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04037"}},"official":null}},{"url":"/paper/dual-policy-distillation","slug":"dual-policy-distillation","title":"Dual Policy Distillation","date":"2020-06-07","arxiv_id":"2006.04061","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/dual-policy-distillation#ran","syntology_url":"https://syntology.ai/paper/2006.04061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04061"}},"official":null}},{"url":"/paper/deployment-efficient-reinforcement-learning","slug":"deployment-efficient-reinforcement-learning","title":"Deployment-Efficient Reinforcement Learning via Model-Based Offline Optimization","date":"2020-06-05","arxiv_id":"2006.03647","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deployment-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03647"}},"official":{"repos":["matsuolab/BREMEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-hard-ai-planning-instances-using","slug":"solving-hard-ai-planning-instances-using","title":"Solving Hard AI Planning Instances Using Curriculum-Driven Deep Reinforcement Learning","date":"2020-06-04","arxiv_id":"2006.02689","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/solving-hard-ai-planning-instances-using#ran","syntology_url":"https://syntology.ai/paper/2006.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.02689"}},"official":null}},{"url":"/paper/visual-transfer-for-reinforcement-learning","slug":"visual-transfer-for-reinforcement-learning","title":"Visual Transfer for Reinforcement Learning via Wasserstein Domain Confusion","date":"2020-06-04","arxiv_id":"2006.03465","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-transfer-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.03465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.03465"}},"official":null}},{"url":"/paper/predicting-goal-directed-human-attention","slug":"predicting-goal-directed-human-attention","title":"Predicting Goal-directed Human Attention Using Inverse Reinforcement Learning","date":"2020-05-28","arxiv_id":"2005.14310","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/predicting-goal-directed-human-attention#ran","syntology_url":"https://syntology.ai/paper/2005.14310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14310"}},"official":{"repos":["cvlab-stonybrook/Scanpath_Prediction"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mopo-model-based-offline-policy-optimization","slug":"mopo-model-based-offline-policy-optimization","title":"MOPO: Model-based Offline Policy Optimization","date":"2020-05-27","arxiv_id":"2005.13239","repositories_listed":6,"syntology":{"n":8,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mopo-model-based-offline-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2005.13239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.13239"}},"official":{"repos":["tianheyu927/mopo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/implementation-matters-in-deep-policy","slug":"implementation-matters-in-deep-policy","title":"Implementation Matters in Deep Policy Gradients: A Case Study on PPO and TRPO","date":"2020-05-25","arxiv_id":"2005.12729","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implementation-matters-in-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2005.12729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.12729"}},"official":{"repos":["MadryLab/implementation-matters"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/local-and-global-explanations-of-agent","slug":"local-and-global-explanations-of-agent","title":"Local and Global Explanations of Agent Behavior: Integrating Strategy Summaries with Saliency Maps","date":"2020-05-18","arxiv_id":"2005.08874","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/local-and-global-explanations-of-agent#ran","syntology_url":"https://syntology.ai/paper/2005.08874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08874"}},"official":{"repos":["HuTobias/HIGHLIGHTS-LRP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/context-aware-dynamics-model-for","slug":"context-aware-dynamics-model-for","title":"Context-aware Dynamics Model for Generalization in Model-Based Reinforcement Learning","date":"2020-05-14","arxiv_id":"2005.06800","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-aware-dynamics-model-for#ran","syntology_url":"https://syntology.ai/paper/2005.06800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.06800"}},"official":{"repos":["younggyoseo/CaDM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/generalized-state-dependent-exploration-for","slug":"generalized-state-dependent-exploration-for","title":"Smooth Exploration for Robotic Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05719","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-state-dependent-exploration-for#ran","syntology_url":"https://syntology.ai/paper/2005.05719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05719"}},"official":{"repos":["DLR-RM/stable-baselines3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/morel-model-based-offline-reinforcement","slug":"morel-model-based-offline-reinforcement","title":"MOReL : Model-Based Offline Reinforcement Learning","date":"2020-05-12","arxiv_id":"2005.05951","repositories_listed":2,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/morel-model-based-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2005.05951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05951"}},"official":null}},{"url":"/paper/planning-to-explore-via-self-supervised-world","slug":"planning-to-explore-via-self-supervised-world","title":"Planning to Explore via Self-Supervised World Models","date":"2020-05-12","arxiv_id":"2005.05960","repositories_listed":4,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/planning-to-explore-via-self-supervised-world#ran","syntology_url":"https://syntology.ai/paper/2005.05960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05960"}},"official":{"repos":["ramanans1/plan2explore"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/carl-controllable-agent-with-reinforcement","slug":"carl-controllable-agent-with-reinforcement","title":"CARL: Controllable Agent with Reinforcement Learning for Quadruped Locomotion","date":"2020-05-07","arxiv_id":"2005.03288","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/carl-controllable-agent-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2005.03288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03288"}},"official":{"repos":["inventec-ai-center/carl-siggraph2020"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/curious-hierarchical-actor-critic","slug":"curious-hierarchical-actor-critic","title":"Curious Hierarchical Actor-Critic Reinforcement Learning","date":"2020-05-07","arxiv_id":"2005.03420","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curious-hierarchical-actor-critic#ran","syntology_url":"https://syntology.ai/paper/2005.03420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03420"}},"official":{"repos":["knowledgetechnologyuhh/goal_conditioned_RL_baselines"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-augmented-data","slug":"reinforcement-learning-with-augmented-data","title":"Reinforcement Learning with Augmented Data","date":"2020-04-30","arxiv_id":"2004.14990","repositories_listed":2,"syntology":{"n":24,"n_ran":21,"n_constructed":8,"n_ran_checked":9,"n_instrument":12,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":21,"phrase":"21 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 12 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reinforcement-learning-with-augmented-data#ran","syntology_url":"https://syntology.ai/paper/2004.14990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14990"}},"official":null}},{"url":"/paper/image-augmentation-is-all-you-need","slug":"image-augmentation-is-all-you-need","title":"Image Augmentation Is All You Need: Regularizing Deep Reinforcement Learning from Pixels","date":"2020-04-28","arxiv_id":"2004.13649","repositories_listed":4,"syntology":{"n":10,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-augmentation-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2004.13649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.13649"}},"official":{"repos":["denisyarats/drq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/first-return-then-explore","slug":"first-return-then-explore","title":"First return, then explore","date":"2020-04-27","arxiv_id":"2004.12919","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/first-return-then-explore#ran","syntology_url":"https://syntology.ai/paper/2004.12919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.12919"}},"official":{"repos":["uber-research/go-explore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emergent-real-world-robotic-skills-via","slug":"emergent-real-world-robotic-skills-via","title":"Emergent Real-World Robotic Skills via Unsupervised Off-Policy Reinforcement Learning","date":"2020-04-27","arxiv_id":"2004.12974","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergent-real-world-robotic-skills-via#ran","syntology_url":"https://syntology.ai/paper/2004.12974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.12974"}},"official":null}},{"url":"/paper/chip-placement-with-deep-reinforcement","slug":"chip-placement-with-deep-reinforcement","title":"Chip Placement with Deep Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10746","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chip-placement-with-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.10746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.10746"}},"official":null}},{"url":"/paper/energy-based-imitation-learning","slug":"energy-based-imitation-learning","title":"Energy-Based Imitation Learning","date":"2020-04-20","arxiv_id":"2004.09395","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/energy-based-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2004.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09395"}},"official":{"repos":["apexrl/EBIL-torch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/continual-reinforcement-learning-with-multi","slug":"continual-reinforcement-learning-with-multi","title":"Continual Reinforcement Learning with Multi-Timescale Replay","date":"2020-04-16","arxiv_id":"2004.07530","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/continual-reinforcement-learning-with-multi#ran","syntology_url":"https://syntology.ai/paper/2004.07530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07530"}},"official":{"repos":["ChristosKap/multi_timescale_replay"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-game-theoretic-framework-for-model-based","slug":"a-game-theoretic-framework-for-model-based","title":"A Game Theoretic Framework for Model Based Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07804","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-game-theoretic-framework-for-model-based#ran","syntology_url":"https://syntology.ai/paper/2004.07804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07804"}},"official":null}},{"url":"/paper/datasets-for-data-driven-reinforcement","slug":"datasets-for-data-driven-reinforcement","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07219","repositories_listed":7,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/datasets-for-data-driven-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.07219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07219"}},"official":{"repos":["rail-berkeley/d4rl","rail-berkeley/offline_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/curl-contrastive-unsupervised-representations","slug":"curl-contrastive-unsupervised-representations","title":"CURL: Contrastive Unsupervised Representations for Reinforcement Learning","date":"2020-04-08","arxiv_id":"2004.04136","repositories_listed":7,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curl-contrastive-unsupervised-representations#ran","syntology_url":"https://syntology.ai/paper/2004.04136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.04136"}},"official":{"repos":["MishaLaskin/curl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/guided-dialog-policy-learning-without","slug":"guided-dialog-policy-learning-without","title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop","date":"2020-04-07","arxiv_id":"2004.03267","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-dialog-policy-learning-without#ran","syntology_url":"https://syntology.ai/paper/2004.03267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.03267"}},"official":{"repos":["cszmli/dp-without-adv"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for-2","slug":"multi-agent-reinforcement-learning-for-2","title":"Multi-agent Reinforcement Learning for Networked System Control","date":"2020-04-03","arxiv_id":"2004.01339","repositories_listed":1,"syntology":{"n":16,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":14,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":16,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/2004.01339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.01339"}},"official":{"repos":["cts198859/deeprl_network"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/action-space-shaping-in-deep-reinforcement","slug":"action-space-shaping-in-deep-reinforcement","title":"Action Space Shaping in Deep Reinforcement Learning","date":"2020-04-02","arxiv_id":"2004.00980","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-space-shaping-in-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.00980","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00980"}},"official":{"repos":["Miffyli/rl-action-space-shaping"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-sparse-rewarded-tasks-from-sub","slug":"learning-sparse-rewarded-tasks-from-sub","title":"Learning Sparse Rewarded Tasks from Sub-Optimal Demonstrations","date":"2020-04-01","arxiv_id":"2004.00530","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-sparse-rewarded-tasks-from-sub#ran","syntology_url":"https://syntology.ai/paper/2004.00530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.00530"}},"official":null}},{"url":"/paper/ultrasound-guided-robotic-navigation-with","slug":"ultrasound-guided-robotic-navigation-with","title":"Ultrasound-Guided Robotic Navigation with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13321","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ultrasound-guided-robotic-navigation-with#ran","syntology_url":"https://syntology.ai/paper/2003.13321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13321"}},"official":{"repos":["hhase/spinal-navigation-rl","hhase/sacrum_data-set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/agent57-outperforming-the-atari-human","slug":"agent57-outperforming-the-atari-human","title":"Agent57: Outperforming the Atari Human Benchmark","date":"2020-03-30","arxiv_id":"2003.13350","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agent57-outperforming-the-atari-human#ran","syntology_url":"https://syntology.ai/paper/2003.13350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13350"}},"official":null}},{"url":"/paper/sample-efficient-ensemble-learning-with","slug":"sample-efficient-ensemble-learning-with","title":"Sample Efficient Ensemble Learning with Catalyst.RL","date":"2020-03-29","arxiv_id":"2003.14210","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sample-efficient-ensemble-learning-with#ran","syntology_url":"https://syntology.ai/paper/2003.14210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.14210"}},"official":{"repos":["Scitator/run-skeleton-run-in-3d"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/policy-teaching-via-environment-poisoning","slug":"policy-teaching-via-environment-poisoning","title":"Policy Teaching via Environment Poisoning: Training-time Adversarial Attacks against Reinforcement Learning","date":"2020-03-28","arxiv_id":"2003.12909","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/policy-teaching-via-environment-poisoning#ran","syntology_url":"https://syntology.ai/paper/2003.12909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12909"}},"official":{"repos":["adishs/icml2020_rl-policy-teaching_code"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-3d-shapes-by-reinforcement-learning","slug":"modeling-3d-shapes-by-reinforcement-learning","title":"Modeling 3D Shapes by Reinforcement Learning","date":"2020-03-27","arxiv_id":"2003.12397","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-3d-shapes-by-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.12397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12397"}},"official":null}},{"url":"/paper/fiber-a-platform-for-efficient-development","slug":"fiber-a-platform-for-efficient-development","title":"Fiber: A Platform for Efficient Development and Distributed Training for Reinforcement Learning and Population-Based Methods","date":"2020-03-25","arxiv_id":"2003.11164","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fiber-a-platform-for-efficient-development#ran","syntology_url":"https://syntology.ai/paper/2003.11164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11164"}},"official":null}},{"url":"/paper/an-empirical-investigation-of-the-challenges","slug":"an-empirical-investigation-of-the-challenges","title":"An empirical investigation of the challenges of real-world reinforcement learning","date":"2020-03-24","arxiv_id":"2003.11881","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-empirical-investigation-of-the-challenges#ran","syntology_url":"https://syntology.ai/paper/2003.11881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.11881"}},"official":{"repos":["google-research/realworldrl_suite"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-population-curriculum-for-1","slug":"evolutionary-population-curriculum-for-1","title":"Evolutionary Population Curriculum for Scaling Multi-Agent Reinforcement Learning","date":"2020-03-23","arxiv_id":"2003.10423","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolutionary-population-curriculum-for-1#ran","syntology_url":"https://syntology.ai/paper/2003.10423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10423"}},"official":{"repos":["qian18long/epciclr2020"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhanced-poet-open-ended-reinforcement","slug":"enhanced-poet-open-ended-reinforcement","title":"Enhanced POET: Open-Ended Reinforcement Learning through Unbounded Invention of Learning Challenges and their Solutions","date":"2020-03-19","arxiv_id":"2003.08536","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhanced-poet-open-ended-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.08536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08536"}},"official":{"repos":["uber-research/poet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monotonic-value-function-factorisation-for","slug":"monotonic-value-function-factorisation-for","title":"Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monotonic-value-function-factorisation-for#ran","syntology_url":"https://syntology.ai/paper/2003.08839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08839"}},"official":{"repos":["oxwhirl/pymarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neuroevolution-of-self-interpretable-agents","slug":"neuroevolution-of-self-interpretable-agents","title":"Neuroevolution of Self-Interpretable Agents","date":"2020-03-18","arxiv_id":"2003.08165","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neuroevolution-of-self-interpretable-agents#ran","syntology_url":"https://syntology.ai/paper/2003.08165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.08165"}},"official":null}},{"url":"/paper/self-supervised-discovering-of-causal","slug":"self-supervised-discovering-of-causal","title":"Self-Supervised Discovering of Interpretable Features for Reinforcement Learning","date":"2020-03-16","arxiv_id":"2003.07069","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-discovering-of-causal#ran","syntology_url":"https://syntology.ai/paper/2003.07069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07069"}},"official":{"repos":["shiwj16/SSINet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discor-corrective-feedback-in-reinforcement","slug":"discor-corrective-feedback-in-reinforcement","title":"DisCor: Corrective Feedback in Reinforcement Learning via Distribution Correction","date":"2020-03-16","arxiv_id":"2003.07305","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/discor-corrective-feedback-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2003.07305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07305"}},"official":null}},{"url":"/paper/deep-deterministic-portfolio-optimization","slug":"deep-deterministic-portfolio-optimization","title":"Deep Deterministic Portfolio Optimization","date":"2020-03-13","arxiv_id":"2003.06497","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-deterministic-portfolio-optimization#ran","syntology_url":"https://syntology.ai/paper/2003.06497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06497"}},"official":{"repos":["CFMTech/Deep-RL-for-Portfolio-Optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-reinforcement-learning","slug":"sample-efficient-reinforcement-learning","title":"Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","date":"2020-03-12","arxiv_id":"2003.06066","repositories_listed":3,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sample-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.06066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06066"}},"official":null}},{"url":"/paper/the-minerl-competition-on-sample-efficient-1","slug":"the-minerl-competition-on-sample-efficient-1","title":"Retrospective Analysis of the 2019 MineRL Competition on Sample Efficient Reinforcement Learning","date":"2020-03-10","arxiv_id":"2003.05012","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-minerl-competition-on-sample-efficient-1#ran","syntology_url":"https://syntology.ai/paper/2003.05012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05012"}},"official":null}},{"url":"/paper/stable-policy-optimization-via-off-policy","slug":"stable-policy-optimization-via-off-policy","title":"Stable Policy Optimization via Off-Policy Divergence Regularization","date":"2020-03-09","arxiv_id":"2003.04108","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stable-policy-optimization-via-off-policy#ran","syntology_url":"https://syntology.ai/paper/2003.04108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.04108"}},"official":{"repos":["facebookresearch/ppo-dice"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autophase-juggling-hls-phase-orderings-in","slug":"autophase-juggling-hls-phase-orderings-in","title":"AutoPhase: Juggling HLS Phase Orderings in Random Forests with Deep Reinforcement Learning","date":"2020-03-02","arxiv_id":"2003.00671","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autophase-juggling-hls-phase-orderings-in#ran","syntology_url":"https://syntology.ai/paper/2003.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00671"}},"official":{"repos":["ucb-bar/autophase"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hybrid-stochastic-policy-gradient-algorithm","slug":"a-hybrid-stochastic-policy-gradient-algorithm","title":"A Hybrid Stochastic Policy Gradient Algorithm for Reinforcement Learning","date":"2020-03-01","arxiv_id":"2003.00430","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hybrid-stochastic-policy-gradient-algorithm#ran","syntology_url":"https://syntology.ai/paper/2003.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.00430"}},"official":{"repos":["unc-optimization/ProxHSPGA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimistic-exploration-even-with-a-1","slug":"optimistic-exploration-even-with-a-1","title":"Optimistic Exploration even with a Pessimistic Initialisation","date":"2020-02-26","arxiv_id":"2002.12174","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimistic-exploration-even-with-a-1#ran","syntology_url":"https://syntology.ai/paper/2002.12174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.12174"}},"official":{"repos":["oxwhirl/opiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-avoid-being-eaten-by-a-grue","slug":"how-to-avoid-being-eaten-by-a-grue","title":"How To Avoid Being Eaten By a Grue: Exploration Strategies for Text-Adventure Agents","date":"2020-02-19","arxiv_id":"2002.08795","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-avoid-being-eaten-by-a-grue#ran","syntology_url":"https://syntology.ai/paper/2002.08795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08795"}},"official":{"repos":["rajammanabrolu/Q-BERT"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/generating-automatic-curricula-via-self","slug":"generating-automatic-curricula-via-self","title":"Generating Automatic Curricula via Self-Supervised Active Domain Randomization","date":"2020-02-18","arxiv_id":"2002.07911","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-automatic-curricula-via-self#ran","syntology_url":"https://syntology.ai/paper/2002.07911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.07911"}},"official":null}},{"url":"/paper/control-frequency-adaptation-via-action","slug":"control-frequency-adaptation-via-action","title":"Control Frequency Adaptation via Action Persistence in Batch Reinforcement Learning","date":"2020-02-17","arxiv_id":"2002.06836","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/control-frequency-adaptation-via-action#ran","syntology_url":"https://syntology.ai/paper/2002.06836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06836"}},"official":{"repos":["albertometelli/pfqi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pddlgym-gym-environments-from-pddl-problems","slug":"pddlgym-gym-environments-from-pddl-problems","title":"PDDLGym: Gym Environments from PDDL Problems","date":"2020-02-15","arxiv_id":"2002.06432","repositories_listed":2,"syntology":{"n":11,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/pddlgym-gym-environments-from-pddl-problems#ran","syntology_url":"https://syntology.ai/paper/2002.06432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06432"}},"official":{"repos":["tomsilver/pddlgym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/an-inductive-bias-for-distances-neural-nets-1","slug":"an-inductive-bias-for-distances-neural-nets-1","title":"An Inductive Bias for Distances: Neural Nets that Respect the Triangle Inequality","date":"2020-02-14","arxiv_id":"2002.05825","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-inductive-bias-for-distances-neural-nets-1#ran","syntology_url":"https://syntology.ai/paper/2002.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05825"}},"official":{"repos":["spitis/deepnorms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discrete-action-on-policy-learning-with","slug":"discrete-action-on-policy-learning-with","title":"Discrete Action On-Policy Learning with Action-Value Critic","date":"2020-02-10","arxiv_id":"2002.03534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-action-on-policy-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.03534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03534"}},"official":{"repos":["yuguangyue/CARSM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-portfolio","slug":"reinforcement-learning-based-portfolio","title":"Reinforcement-Learning based Portfolio Management with Augmented Asset Movement Prediction States","date":"2020-02-09","arxiv_id":"2002.05780","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/reinforcement-learning-based-portfolio#ran","syntology_url":"https://syntology.ai/paper/2002.05780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05780"}},"official":null}},{"url":"/paper/multi-type-mean-field-reinforcement-learning","slug":"multi-type-mean-field-reinforcement-learning","title":"Multi Type Mean Field Reinforcement Learning","date":"2020-02-06","arxiv_id":"2002.02513","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-type-mean-field-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2002.02513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02513"}},"official":{"repos":["BorealisAI/mtmfrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/one-shot-bayes-opt-with-probabilistic","slug":"one-shot-bayes-opt-with-probabilistic","title":"Provably Efficient Online Hyperparameter Optimization with Population-Based Bandits","date":"2020-02-06","arxiv_id":"2002.02518","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/one-shot-bayes-opt-with-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2002.02518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02518"}},"official":{"repos":["jparkerholder/PB2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-reinforcement-learning-framework-for-time","slug":"a-reinforcement-learning-framework-for-time","title":"Dynamic Causal Effects Evaluation in A/B Testing with a Reinforcement Learning Framework","date":"2020-02-05","arxiv_id":"2002.01711","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-reinforcement-learning-framework-for-time#ran","syntology_url":"https://syntology.ai/paper/2002.01711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01711"}},"official":{"repos":["callmespring/causalrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-diversity-in-population-based","slug":"effective-diversity-in-population-based","title":"Effective Diversity in Population Based Reinforcement Learning","date":"2020-02-03","arxiv_id":"2002.00632","repositories_listed":2,"syntology":{"n":12,"n_ran":10,"n_constructed":6,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":9,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/effective-diversity-in-population-based#ran","syntology_url":"https://syntology.ai/paper/2002.00632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00632"}},"official":{"repos":["jparkerholder/DvD_ES"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/integrating-deep-reinforcement-learning-with","slug":"integrating-deep-reinforcement-learning-with","title":"Integrating Deep Reinforcement Learning with Model-based Path Planners for Automated Driving","date":"2020-02-02","arxiv_id":"2002.00434","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/integrating-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2002.00434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.00434"}},"official":{"repos":["Ekim-Yurtsever/Hybrid-DeepRL-Automated-Driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-the-systematic-reporting-of-the","slug":"towards-the-systematic-reporting-of-the","title":"Towards the Systematic Reporting of the Energy and Carbon Footprints of Machine Learning","date":"2020-01-31","arxiv_id":"2002.05651","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/towards-the-systematic-reporting-of-the#ran","syntology_url":"https://syntology.ai/paper/2002.05651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.05651"}},"official":{"repos":["Breakend/ClimateChangeFromMachineLearningResearch","Breakend/experiment-impact-tracker"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pcgrl-procedural-content-generation-via","slug":"pcgrl-procedural-content-generation-via","title":"PCGRL: Procedural Content Generation via Reinforcement Learning","date":"2020-01-24","arxiv_id":"2001.09212","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pcgrl-procedural-content-generation-via#ran","syntology_url":"https://syntology.ai/paper/2001.09212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.09212"}},"official":{"repos":["amidos2006/gym-pcgrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/interpretable-end-to-end-urban-autonomous","slug":"interpretable-end-to-end-urban-autonomous","title":"Interpretable End-to-end Urban Autonomous Driving with Latent Deep Reinforcement Learning","date":"2020-01-23","arxiv_id":"2001.08726","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/interpretable-end-to-end-urban-autonomous#ran","syntology_url":"https://syntology.ai/paper/2001.08726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08726"}},"official":{"repos":["cjy1992/interp-e2e-driving"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gradient-surgery-for-multi-task-learning-1","slug":"gradient-surgery-for-multi-task-learning-1","title":"Gradient Surgery for Multi-Task Learning","date":"2020-01-19","arxiv_id":"2001.06782","repositories_listed":18,"syntology":{"n":10,"n_ran":7,"n_constructed":5,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gradient-surgery-for-multi-task-learning-1#ran","syntology_url":"https://syntology.ai/paper/2001.06782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06782"}},"official":{"repos":["tianheyu927/PCGrad"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/tree-structured-policy-based-progressive","slug":"tree-structured-policy-based-progressive","title":"Tree-Structured Policy based Progressive Reinforcement Learning for Temporally Language Grounding in Video","date":"2020-01-18","arxiv_id":"2001.06680","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tree-structured-policy-based-progressive#ran","syntology_url":"https://syntology.ai/paper/2001.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06680"}},"official":{"repos":["WuJie1010/TSP-PRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/popcorn-partially-observed-prediction","slug":"popcorn-partially-observed-prediction","title":"POPCORN: Partially Observed Prediction COnstrained ReiNforcement Learning","date":"2020-01-13","arxiv_id":"2001.04032","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/popcorn-partially-observed-prediction#ran","syntology_url":"https://syntology.ai/paper/2001.04032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04032"}},"official":{"repos":["dtak/POPCORN-POMDP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/statistical-inference-of-the-value-function","slug":"statistical-inference-of-the-value-function","title":"Statistical Inference of the Value Function for Reinforcement Learning in Infinite Horizon Settings","date":"2020-01-13","arxiv_id":"2001.04515","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/statistical-inference-of-the-value-function#ran","syntology_url":"https://syntology.ai/paper/2001.04515","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04515"}},"official":{"repos":["shengzhang37/SAVE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/making-sense-of-reinforcement-learning-and-1","slug":"making-sense-of-reinforcement-learning-and-1","title":"Making Sense of Reinforcement Learning and Probabilistic Inference","date":"2020-01-03","arxiv_id":"2001.00805","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-sense-of-reinforcement-learning-and-1#ran","syntology_url":"https://syntology.ai/paper/2001.00805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00805"}},"official":null}},{"url":"/paper/meta-reinforcement-learning-with-autonomous-1","slug":"meta-reinforcement-learning-with-autonomous-1","title":"Meta Reinforcement Learning with Autonomous Inference of Subtask Dependencies","date":"2020-01-01","arxiv_id":"2001.00248","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-reinforcement-learning-with-autonomous-1#ran","syntology_url":"https://syntology.ai/paper/2001.00248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00248"}},"official":{"repos":["srsohn/msgi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-conditioned-policies","slug":"reward-conditioned-policies","title":"Reward-Conditioned Policies","date":"2019-12-31","arxiv_id":"1912.13465","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-conditioned-policies#ran","syntology_url":"https://syntology.ai/paper/1912.13465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.13465"}},"official":null}},{"url":"/paper/pac-confidence-sets-for-deep-neural-networks-1","slug":"pac-confidence-sets-for-deep-neural-networks-1","title":"PAC Confidence Sets for Deep Neural Networks via Calibrated Prediction","date":"2019-12-31","arxiv_id":"2001.00106","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/pac-confidence-sets-for-deep-neural-networks-1#ran","syntology_url":"https://syntology.ai/paper/2001.00106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00106"}},"official":{"repos":["sangdon/PAC-confidence-set"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1","slug":"slm-lab-a-comprehensive-benchmark-and-modular-1","title":"SLM Lab: A Comprehensive Benchmark and Modular Software Framework for Reproducible Deep Reinforcement Learning","date":"2019-12-28","arxiv_id":"1912.12482","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/slm-lab-a-comprehensive-benchmark-and-modular-1#ran","syntology_url":"https://syntology.ai/paper/1912.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12482"}},"official":{"repos":["kengz/SLM-Lab"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-practical-multi-object-manipulation","slug":"towards-practical-multi-object-manipulation","title":"Towards Practical Multi-Object Manipulation using Relational Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.11032","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-practical-multi-object-manipulation#ran","syntology_url":"https://syntology.ai/paper/1912.11032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11032"}},"official":null}},{"url":"/paper/discrete-and-continuous-action-representation","slug":"discrete-and-continuous-action-representation","title":"Discrete and Continuous Action Representation for Practical RL in Video Games","date":"2019-12-23","arxiv_id":"1912.11077","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discrete-and-continuous-action-representation#ran","syntology_url":"https://syntology.ai/paper/1912.11077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11077"}},"official":null}}],"record_sha256":"b7cd3e7b2bdb8fdec956dbc7127c3efb06131c1b39549ad26b38afb3374cda32","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}