{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/ran/9","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":9,"pages_in_order":12,"rows_per_page":100,"rows":[801,900],"of":1175,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning/papers/ran/1","prev":"/task/reinforcement-learning/papers/ran/8","next":"/task/reinforcement-learning/papers/ran/10","papers":[{"url":"/paper/contextual-imagined-goals-for-self-supervised","slug":"contextual-imagined-goals-for-self-supervised","title":"Contextual Imagined Goals for Self-Supervised Robotic Learning","date":"2019-10-23","arxiv_id":"1910.11670","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-imagined-goals-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/1910.11670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11670"}},"official":null}},{"url":"/paper/regularization-matters-in-policy-optimization-1","slug":"regularization-matters-in-policy-optimization-1","title":"Regularization Matters in Policy Optimization","date":"2019-10-21","arxiv_id":"1910.09191","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularization-matters-in-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/1910.09191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09191"}},"official":{"repos":["xuanlinli17/iclr2021_rlreg","xuanlinli17/po-rl-regularization"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/learning-to-map-natural-language-instructions","slug":"learning-to-map-natural-language-instructions","title":"Learning to Map Natural Language Instructions to Physical Quadcopter Control using Simulated Flight","date":"2019-10-21","arxiv_id":"1910.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-map-natural-language-instructions#ran","syntology_url":"https://syntology.ai/paper/1910.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09664"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rtfm-generalising-to-novel-environment","slug":"rtfm-generalising-to-novel-environment","title":"RTFM: Generalising to Novel Environment Dynamics via Reading","date":"2019-10-18","arxiv_id":"1910.08210","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rtfm-generalising-to-novel-environment#ran","syntology_url":"https://syntology.ai/paper/1910.08210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.08210"}},"official":null}},{"url":"/paper/soft-actor-critic-for-discrete-action","slug":"soft-actor-critic-for-discrete-action","title":"Soft Actor-Critic for Discrete Action Settings","date":"2019-10-16","arxiv_id":"1910.07207","repositories_listed":13,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":6,"n_honours":2,"n_violates":1,"n_no_contract":9,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/soft-actor-critic-for-discrete-action#ran","syntology_url":"https://syntology.ai/paper/1910.07207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07207"}},"official":{"repos":["p-christ/Deep-Reinforcement-Learning-Algorithms-with-PyTorch"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/teacher-algorithms-for-curriculum-learning-of","slug":"teacher-algorithms-for-curriculum-learning-of","title":"Teacher algorithms for curriculum learning of Deep RL in continuously parameterized environments","date":"2019-10-16","arxiv_id":"1910.07224","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teacher-algorithms-for-curriculum-learning-of#ran","syntology_url":"https://syntology.ai/paper/1910.07224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07224"}},"official":{"repos":["flowersteam/teachDeepRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-meets-graph","slug":"deep-reinforcement-learning-meets-graph","title":"Deep Reinforcement Learning meets Graph Neural Networks: exploring a routing optimization use case","date":"2019-10-16","arxiv_id":"1910.07421","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-meets-graph#ran","syntology_url":"https://syntology.ai/paper/1910.07421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07421"}},"official":{"repos":["knowledgedefinednetworking/DRL-GNN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/maven-multi-agent-variational-exploration","slug":"maven-multi-agent-variational-exploration","title":"MAVEN: Multi-Agent Variational Exploration","date":"2019-10-16","arxiv_id":"1910.07483","repositories_listed":4,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/maven-multi-agent-variational-exploration#ran","syntology_url":"https://syntology.ai/paper/1910.07483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07483"}},"official":{"repos":["AnujMahajanOxf/MAVEN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/seed-rl-scalable-and-efficient-deep-rl-with-1","slug":"seed-rl-scalable-and-efficient-deep-rl-with-1","title":"SEED RL: Scalable and Efficient Deep-RL with Accelerated Central Inference","date":"2019-10-15","arxiv_id":"1910.06591","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seed-rl-scalable-and-efficient-deep-rl-with-1#ran","syntology_url":"https://syntology.ai/paper/1910.06591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06591"}},"official":{"repos":["google-research/seed_rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformers-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1910.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06764"}},"official":null}},{"url":"/paper/influence-based-multi-agent-exploration","slug":"influence-based-multi-agent-exploration","title":"Influence-Based Multi-Agent Exploration","date":"2019-10-12","arxiv_id":"1910.05512","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/influence-based-multi-agent-exploration#ran","syntology_url":"https://syntology.ai/paper/1910.05512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05512"}},"official":{"repos":["TonghanWang/EITI-EDTI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-randomization-technique-for-1","slug":"a-simple-randomization-technique-for-1","title":"Network Randomization: A Simple Technique for Generalization in Deep Reinforcement Learning","date":"2019-10-11","arxiv_id":"1910.05396","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-randomization-technique-for-1#ran","syntology_url":"https://syntology.ai/paper/1910.05396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05396"}},"official":{"repos":["pokaxpoka/netrand"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlcard-a-toolkit-for-reinforcement-learning","slug":"rlcard-a-toolkit-for-reinforcement-learning","title":"RLCard: A Toolkit for Reinforcement Learning in Card Games","date":"2019-10-10","arxiv_id":"1910.04376","repositories_listed":9,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rlcard-a-toolkit-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.04376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04376"}},"official":{"repos":["datamllab/rlcard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/assistive-gym-a-physics-simulation-framework","slug":"assistive-gym-a-physics-simulation-framework","title":"Assistive Gym: A Physics Simulation Framework for Assistive Robotics","date":"2019-10-10","arxiv_id":"1910.04700","repositories_listed":4,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assistive-gym-a-physics-simulation-framework#ran","syntology_url":"https://syntology.ai/paper/1910.04700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04700"}},"official":{"repos":["Healthcare-Robotics/assistive-gym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/torchbeast-a-pytorch-platform-for-distributed","slug":"torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/torchbeast-a-pytorch-platform-for-distributed#ran","syntology_url":"https://syntology.ai/paper/1910.03552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.03552"}},"official":{"repos":["heiner/scalable_agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/self-paced-contextual-reinforcement-learning","slug":"self-paced-contextual-reinforcement-learning","title":"Self-Paced Contextual Reinforcement Learning","date":"2019-10-07","arxiv_id":"1910.02826","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-paced-contextual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.02826"}},"official":{"repos":["psclklnk/self-paced-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reducing-overestimation-bias-in-multi-agent","slug":"reducing-overestimation-bias-in-multi-agent","title":"Reducing Overestimation Bias in Multi-Agent Domains Using Double Centralized Critics","date":"2019-10-03","arxiv_id":"1910.01465","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/reducing-overestimation-bias-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1910.01465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01465"}},"official":{"repos":["JohannesAck/MATD3implementation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantized-reinforcement-learning-quarl#ran","syntology_url":"https://syntology.ai/paper/1910.01055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01055"}},"official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-sample-efficiency-in-model-free-1","slug":"improving-sample-efficiency-in-model-free-1","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","date":"2019-10-02","arxiv_id":"1910.01741","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-sample-efficiency-in-model-free-1#ran","syntology_url":"https://syntology.ai/paper/1910.01741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01741"}},"official":{"repos":["denisyarats/pytorch_sac_ae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/hamiltonian-generative-networks","slug":"hamiltonian-generative-networks","title":"Hamiltonian Generative Networks","date":"2019-09-30","arxiv_id":"1909.13789","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hamiltonian-generative-networks#ran","syntology_url":"https://syntology.ai/paper/1909.13789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.13789"}},"official":null}},{"url":"/paper/meta-q-learning","slug":"meta-q-learning","title":"Meta-Q-Learning","date":"2019-09-30","arxiv_id":"1910.00125","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-q-learning#ran","syntology_url":"https://syntology.ai/paper/1910.00125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00125"}},"official":{"repos":["amazon-research/meta-q-learning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy","slug":"v-mpo-on-policy-maximum-a-posteriori-policy","title":"V-MPO: On-Policy Maximum a Posteriori Policy Optimization for Discrete and Continuous Control","date":"2019-09-26","arxiv_id":"1909.12238","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-mpo-on-policy-maximum-a-posteriori-policy#ran","syntology_url":"https://syntology.ai/paper/1909.12238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12238"}},"official":null}},{"url":"/paper/harnessing-structures-for-value-based","slug":"harnessing-structures-for-value-based","title":"Harnessing Structures for Value-Based Planning and Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12255","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-structures-for-value-based#ran","syntology_url":"https://syntology.ai/paper/1909.12255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12255"}},"official":{"repos":["YyzHarry/SV-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-valuation-using-reinforcement-learning","slug":"data-valuation-using-reinforcement-learning","title":"Data Valuation using Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11671","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/data-valuation-using-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1909.11671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11671"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fine-tuning-language-models-from-human","slug":"fine-tuning-language-models-from-human","title":"Fine-Tuning Language Models from Human Preferences","date":"2019-09-18","arxiv_id":"1909.08593","repositories_listed":9,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-tuning-language-models-from-human#ran","syntology_url":"https://syntology.ai/paper/1909.08593","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.08593"}},"official":{"repos":["openai/lm-human-preferences"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/emergent-tool-use-from-multi-agent","slug":"emergent-tool-use-from-multi-agent","title":"Emergent Tool Use From Multi-Agent Autocurricula","date":"2019-09-17","arxiv_id":"1909.07528","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergent-tool-use-from-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1909.07528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07528"}},"official":{"repos":["openai/multi-agent-emergence-environments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mdp-playground-meta-features-in-reinforcement","slug":"mdp-playground-meta-features-in-reinforcement","title":"MDP Playground: An Analysis and Debug Testbed for Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.07750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-playground-meta-features-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1909.07750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07750"}},"official":{"repos":["automl/mdp-playground"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recsim-a-configurable-simulation-platform-for","slug":"recsim-a-configurable-simulation-platform-for","title":"RecSim: A Configurable Simulation Platform for Recommender Systems","date":"2019-09-11","arxiv_id":"1909.04847","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recsim-a-configurable-simulation-platform-for#ran","syntology_url":"https://syntology.ai/paper/1909.04847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04847"}},"official":{"repos":["google-research/recsim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-temporal-logic","slug":"reinforcement-learning-for-temporal-logic","title":"Reinforcement Learning for Temporal Logic Control Synthesis with Probabilistic Satisfaction Guarantees","date":"2019-09-11","arxiv_id":"1909.05304","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-temporal-logic#ran","syntology_url":"https://syntology.ai/paper/1909.05304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05304"}},"official":{"repos":["grockious/lcrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-a-good-story-designing-composite","slug":"what-makes-a-good-story-designing-composite","title":"What Makes A Good Story? Designing Composite Rewards for Visual Storytelling","date":"2019-09-11","arxiv_id":"1909.05316","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-makes-a-good-story-designing-composite#ran","syntology_url":"https://syntology.ai/paper/1909.05316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05316"}},"official":{"repos":["JunjieHu/ReCo-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-reproducibility-by-evaluating","slug":"a-survey-on-reproducibility-by-evaluating","title":"A Survey on Reproducibility by Evaluating Deep Reinforcement Learning Algorithms on Real-World Robots","date":"2019-09-09","arxiv_id":"1909.03772","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-survey-on-reproducibility-by-evaluating#ran","syntology_url":"https://syntology.ai/paper/1909.03772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.03772"}},"official":{"repos":["dti-research/SenseActExperiments"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploratory-combinatorial-optimization-with","slug":"exploratory-combinatorial-optimization-with","title":"Exploratory Combinatorial Optimization with Reinforcement Learning","date":"2019-09-09","arxiv_id":"1909.04063","repositories_listed":2,"syntology":{"n":20,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/exploratory-combinatorial-optimization-with#ran","syntology_url":"https://syntology.ai/paper/1909.04063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04063"}},"official":{"repos":["tomdbar/eco-dqn"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/classification-with-costly-features-as-a","slug":"classification-with-costly-features-as-a","title":"Classification with Costly Features as a Sequential Decision-Making Problem","date":"2019-09-05","arxiv_id":"1909.02564","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/classification-with-costly-features-as-a#ran","syntology_url":"https://syntology.ai/paper/1909.02564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02564"}},"official":{"repos":["jaromiru/cwcf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/answers-unite-unsupervised-metrics-for","slug":"answers-unite-unsupervised-metrics-for","title":"Answers Unite! Unsupervised Metrics for Reinforced Summarization Models","date":"2019-09-04","arxiv_id":"1909.01610","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answers-unite-unsupervised-metrics-for#ran","syntology_url":"https://syntology.ai/paper/1909.01610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01610"}},"official":{"repos":["recitalAI/summa-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-dynamic-context-augmentation-for","slug":"learning-dynamic-context-augmentation-for","title":"Learning Dynamic Context Augmentation for Global Entity Linking","date":"2019-09-04","arxiv_id":"1909.02117","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-dynamic-context-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/1909.02117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02117"}},"official":{"repos":["YoungXiyuan/DCA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/no-press-diplomacy-modeling-multi-agent","slug":"no-press-diplomacy-modeling-multi-agent","title":"No Press Diplomacy: Modeling Multi-Agent Gameplay","date":"2019-09-04","arxiv_id":"1909.02128","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/no-press-diplomacy-modeling-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1909.02128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02128"}},"official":{"repos":["diplomacy/research"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/better-rewards-yield-better-summaries","slug":"better-rewards-yield-better-summaries","title":"Better Rewards Yield Better Summaries: Learning to Summarise Without References","date":"2019-09-03","arxiv_id":"1909.01214","repositories_listed":2,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/better-rewards-yield-better-summaries#ran","syntology_url":"https://syntology.ai/paper/1909.01214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01214"}},"official":{"repos":["yg211/summary-reward-no-reference"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/rlpyt-a-research-code-base-for-deep","slug":"rlpyt-a-research-code-base-for-deep","title":"rlpyt: A Research Code Base for Deep Reinforcement Learning in PyTorch","date":"2019-09-03","arxiv_id":"1909.01500","repositories_listed":9,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rlpyt-a-research-code-base-for-deep#ran","syntology_url":"https://syntology.ai/paper/1909.01500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01500"}},"official":{"repos":["astooke/rlpyt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/meta-learning-with-warped-gradient-descent","slug":"meta-learning-with-warped-gradient-descent","title":"Meta-Learning with Warped Gradient Descent","date":"2019-08-30","arxiv_id":"1909.00025","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/meta-learning-with-warped-gradient-descent#ran","syntology_url":"https://syntology.ai/paper/1909.00025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.00025"}},"official":{"repos":["flennerhag/warpgrad"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-comparison-on-imitation-learning","slug":"an-empirical-comparison-on-imitation-learning","title":"An Empirical Comparison on Imitation Learning and Reinforcement Learning for Paraphrase Generation","date":"2019-08-28","arxiv_id":"1908.10835","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-empirical-comparison-on-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1908.10835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10835"}},"official":{"repos":["ddddwy/Reinforce-Paraphrase-Generation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-language-learning-by-question","slug":"interactive-language-learning-by-question","title":"Interactive Language Learning by Question Answering","date":"2019-08-28","arxiv_id":"1908.10909","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interactive-language-learning-by-question#ran","syntology_url":"https://syntology.ai/paper/1908.10909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10909"}},"official":{"repos":["xingdi-eric-yuan/qait_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/openspiel-a-framework-for-reinforcement","slug":"openspiel-a-framework-for-reinforcement","title":"OpenSpiel: A Framework for Reinforcement Learning in Games","date":"2019-08-26","arxiv_id":"1908.09453","repositories_listed":16,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openspiel-a-framework-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1908.09453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.09453"}},"official":{"repos":["deepmind/open_spiel"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/a-generalized-algorithm-for-multi-objective","slug":"a-generalized-algorithm-for-multi-objective","title":"A Generalized Algorithm for Multi-Objective Reinforcement Learning and Policy Adaptation","date":"2019-08-21","arxiv_id":"1908.08342","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-generalized-algorithm-for-multi-objective#ran","syntology_url":"https://syntology.ai/paper/1908.08342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.08342"}},"official":{"repos":["RunzheYang/MORL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-in-world-earth","slug":"deep-reinforcement-learning-in-world-earth","title":"Deep reinforcement learning in World-Earth system models to discover sustainable management strategies","date":"2019-08-15","arxiv_id":"1908.05567","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-in-world-earth#ran","syntology_url":"https://syntology.ai/paper/1908.05567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.05567"}},"official":{"repos":["fstrnad/pyDRLinWESM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-graph-to","slug":"reinforcement-learning-based-graph-to","title":"Reinforcement Learning Based Graph-to-Sequence Model for Natural Question Generation","date":"2019-08-14","arxiv_id":"1908.04942","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-based-graph-to#ran","syntology_url":"https://syntology.ai/paper/1908.04942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.04942"}},"official":{"repos":["hugochan/RL-based-Graph2Seq-for-NQG"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/behaviour-suite-for-reinforcement-learning","slug":"behaviour-suite-for-reinforcement-learning","title":"Behaviour Suite for Reinforcement Learning","date":"2019-08-09","arxiv_id":"1908.03568","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behaviour-suite-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1908.03568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.03568"}},"official":{"repos":["deepmind/bsuite"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/dueling-posterior-sampling-for-preference","slug":"dueling-posterior-sampling-for-preference","title":"Dueling Posterior Sampling for Preference-Based Reinforcement Learning","date":"2019-08-04","arxiv_id":"1908.01289","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dueling-posterior-sampling-for-preference#ran","syntology_url":"https://syntology.ai/paper/1908.01289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.01289"}},"official":{"repos":["ernovoseller/DuelingPosteriorSampling"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-learning-for-efficient-reinforcement","slug":"reward-learning-for-efficient-reinforcement","title":"Reward Learning for Efficient Reinforcement Learning in Extractive Document Summarisation","date":"2019-07-30","arxiv_id":"1907.12894","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-learning-for-efficient-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.12894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.12894"}},"official":{"repos":["UKPLab/ijcai2019-relis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/google-research-football-a-novel","slug":"google-research-football-a-novel","title":"Google Research Football: A Novel Reinforcement Learning Environment","date":"2019-07-25","arxiv_id":"1907.11180","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/google-research-football-a-novel#ran","syntology_url":"https://syntology.ai/paper/1907.11180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.11180"}},"official":{"repos":["google-research/football"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/characterizing-attacks-on-deep-reinforcement","slug":"characterizing-attacks-on-deep-reinforcement","title":"Characterizing Attacks on Deep Reinforcement Learning","date":"2019-07-21","arxiv_id":"1907.09470","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/characterizing-attacks-on-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.09470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.09470"}},"official":null}},{"url":"/paper/gpu-accelerated-atari-emulation-for","slug":"gpu-accelerated-atari-emulation-for","title":"Accelerating Reinforcement Learning through GPU Atari Emulation","date":"2019-07-19","arxiv_id":"1907.08467","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpu-accelerated-atari-emulation-for#ran","syntology_url":"https://syntology.ai/paper/1907.08467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.08467"}},"official":{"repos":["NVLABs/cule"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-policy-improvement-with-soft-baseline","slug":"safe-policy-improvement-with-soft-baseline","title":"Safe Policy Improvement with Soft Baseline Bootstrapping","date":"2019-07-11","arxiv_id":"1907.05079","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-policy-improvement-with-soft-baseline#ran","syntology_url":"https://syntology.ai/paper/1907.05079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.05079"}},"official":{"repos":["RomainLaroche/SPIBB","rems75/SPIBB-DQN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethink-global-reward-game-and-credit","slug":"rethink-global-reward-game-and-credit","title":"Shapley Q-value: A Local Reward Approach to Solve Global Reward Games","date":"2019-07-11","arxiv_id":"1907.05707","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/rethink-global-reward-game-and-credit#ran","syntology_url":"https://syntology.ai/paper/1907.05707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.05707"}},"official":{"repos":["hsvgbkhgbv/SQDDPG"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-policies-for-social-network","slug":"learning-policies-for-social-network","title":"Influence maximization in unknown social networks: Learning Policies for Effective Graph Sampling","date":"2019-07-08","arxiv_id":"1907.11625","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-policies-for-social-network#ran","syntology_url":"https://syntology.ai/paper/1907.11625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.11625"}},"official":{"repos":["kage08/graph_sample_rl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/attentive-multi-task-deep-reinforcement","slug":"attentive-multi-task-deep-reinforcement","title":"Attentive Multi-Task Deep Reinforcement Learning","date":"2019-07-05","arxiv_id":"1907.02874","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/attentive-multi-task-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.02874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.02874"}},"official":{"repos":["braemt/attentive-multi-task-deep-reinforcement-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-q-improvement-reinforcement","slug":"conservative-q-improvement-reinforcement","title":"Conservative Q-Improvement: Reinforcement Learning for an Interpretable Decision-Tree Policy","date":"2019-07-02","arxiv_id":"1907.01180","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conservative-q-improvement-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.01180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01180"}},"official":null}},{"url":"/paper/dynamics-aware-unsupervised-discovery-of","slug":"dynamics-aware-unsupervised-discovery-of","title":"Dynamics-Aware Unsupervised Discovery of Skills","date":"2019-07-02","arxiv_id":"1907.01657","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamics-aware-unsupervised-discovery-of#ran","syntology_url":"https://syntology.ai/paper/1907.01657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01657"}},"official":{"repos":["google-research/dads"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stochastic-latent-actor-critic-deep","slug":"stochastic-latent-actor-critic-deep","title":"Stochastic Latent Actor-Critic: Deep Reinforcement Learning with a Latent Variable Model","date":"2019-07-01","arxiv_id":"1907.00953","repositories_listed":9,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stochastic-latent-actor-critic-deep#ran","syntology_url":"https://syntology.ai/paper/1907.00953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.00953"}},"official":null}},{"url":"/paper/way-off-policy-batch-deep-reinforcement","slug":"way-off-policy-batch-deep-reinforcement","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","date":"2019-06-30","arxiv_id":"1907.00456","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/way-off-policy-batch-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.00456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.00456"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyp-rl-hyperparameter-optimization-by","slug":"hyp-rl-hyperparameter-optimization-by","title":"Hyp-RL : Hyperparameter Optimization by Reinforcement Learning","date":"2019-06-27","arxiv_id":"1906.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyp-rl-hyperparameter-optimization-by#ran","syntology_url":"https://syntology.ai/paper/1906.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11527"}},"official":{"repos":["hadijomaa/HypRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-empathic-deep-q-learning","slug":"towards-empathic-deep-q-learning","title":"Towards Empathic Deep Q-Learning","date":"2019-06-26","arxiv_id":"1906.10918","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-empathic-deep-q-learning#ran","syntology_url":"https://syntology.ai/paper/1906.10918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.10918"}},"official":{"repos":["bartbussmann/EmpathicDQN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monte-carlo-gradient-estimation-in-machine","slug":"monte-carlo-gradient-estimation-in-machine","title":"Monte Carlo Gradient Estimation in Machine Learning","date":"2019-06-25","arxiv_id":"1906.10652","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/monte-carlo-gradient-estimation-in-machine#ran","syntology_url":"https://syntology.ai/paper/1906.10652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.10652"}},"official":{"repos":["deepmind/mc_gradients"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ranking-policy-gradient","slug":"ranking-policy-gradient","title":"Ranking Policy Gradient","date":"2019-06-24","arxiv_id":"1906.09674","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ranking-policy-gradient#ran","syntology_url":"https://syntology.ai/paper/1906.09674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09674"}},"official":{"repos":["illidanlab/rpg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proximal-distilled-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09807"}},"official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-convex","slug":"reinforcement-learning-with-convex","title":"Reinforcement Learning with Convex Constraints","date":"2019-06-21","arxiv_id":"1906.09323","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-convex#ran","syntology_url":"https://syntology.ai/paper/1906.09323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09323"}},"official":{"repos":["xkianteb/ApproPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-trust-your-model-model-based-policy","slug":"when-to-trust-your-model-model-based-policy","title":"When to Trust Your Model: Model-Based Policy Optimization","date":"2019-06-19","arxiv_id":"1906.08253","repositories_listed":11,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-to-trust-your-model-model-based-policy#ran","syntology_url":"https://syntology.ai/paper/1906.08253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.08253"}},"official":{"repos":["JannerM/mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/unsupervised-learning-of-object-keypoints-for","slug":"unsupervised-learning-of-object-keypoints-for","title":"Unsupervised Learning of Object Keypoints for Perception and Control","date":"2019-06-19","arxiv_id":"1906.11883","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-learning-of-object-keypoints-for#ran","syntology_url":"https://syntology.ai/paper/1906.11883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11883"}},"official":{"repos":["deepmind/deepmind-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/language-as-an-abstraction-for-hierarchical","slug":"language-as-an-abstraction-for-hierarchical","title":"Language as an Abstraction for Hierarchical Deep Reinforcement Learning","date":"2019-06-18","arxiv_id":"1906.07343","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-as-an-abstraction-for-hierarchical#ran","syntology_url":"https://syntology.ai/paper/1906.07343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.07343"}},"official":{"repos":["google-research/clevr_robot_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-conditioned-imitation-learning","slug":"goal-conditioned-imitation-learning","title":"Goal-conditioned Imitation Learning","date":"2019-06-13","arxiv_id":"1906.05838","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/goal-conditioned-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1906.05838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05838"}},"official":{"repos":["dingyiming0427/goalgail"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-industrial","slug":"deep-reinforcement-learning-for-industrial","title":"Deep Reinforcement Learning for Industrial Insertion Tasks with Visual Inputs and Natural Rewards","date":"2019-06-13","arxiv_id":"1906.05841","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-for-industrial#ran","syntology_url":"https://syntology.ai/paper/1906.05841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05841"}},"official":null}},{"url":"/paper/nas-fcos-fast-neural-architecture-search-for","slug":"nas-fcos-fast-neural-architecture-search-for","title":"NAS-FCOS: Fast Neural Architecture Search for Object Detection","date":"2019-06-11","arxiv_id":"1906.04423","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nas-fcos-fast-neural-architecture-search-for#ran","syntology_url":"https://syntology.ai/paper/1906.04423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04423"}},"official":null}},{"url":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-soft-actor-critic-emphasizing-recent#ran","syntology_url":"https://syntology.ai/paper/1906.04009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04009"}},"official":null}},{"url":"/paper/neural-keyphrase-generation-via-reinforcement","slug":"neural-keyphrase-generation-via-reinforcement","title":"Neural Keyphrase Generation via Reinforcement Learning with Adaptive Rewards","date":"2019-06-10","arxiv_id":"1906.04106","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-keyphrase-generation-via-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04106"}},"official":{"repos":["kenchan0226/keyphrase-generation-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/svrg-for-policy-evaluation-with-fewer","slug":"svrg-for-policy-evaluation-with-fewer","title":"SVRG for Policy Evaluation with Fewer Gradient Evaluations","date":"2019-06-09","arxiv_id":"1906.03704","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/svrg-for-policy-evaluation-with-fewer#ran","syntology_url":"https://syntology.ai/paper/1906.03704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.03704"}},"official":null}},{"url":"/paper/improving-exploration-in-soft-actor-critic","slug":"improving-exploration-in-soft-actor-critic","title":"Improving Exploration in Soft-Actor-Critic with Normalizing Flows Policies","date":"2019-06-06","arxiv_id":"1906.02771","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-exploration-in-soft-actor-critic#ran","syntology_url":"https://syntology.ai/paper/1906.02771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.02771"}},"official":{"repos":["joeybose/FloRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-transferable-cooperative-behavior-in","slug":"learning-transferable-cooperative-behavior-in","title":"Learning Transferable Cooperative Behavior in Multi-Agent Teams","date":"2019-06-04","arxiv_id":"1906.01202","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-transferable-cooperative-behavior-in#ran","syntology_url":"https://syntology.ai/paper/1906.01202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01202"}},"official":{"repos":["sumitsk/matrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-with-low-complexity#ran","syntology_url":"https://syntology.ai/paper/1906.01695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01695"}},"official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/190600949","slug":"190600949","title":"Stabilizing Off-Policy Q-Learning via Bootstrapping Error Reduction","date":"2019-06-03","arxiv_id":"1906.00949","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/190600949#ran","syntology_url":"https://syntology.ai/paper/1906.00949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.00949"}},"official":null}},{"url":"/paper/extending-deep-model-predictive-control-with","slug":"extending-deep-model-predictive-control-with","title":"Safety Augmented Value Estimation from Demonstrations (SAVED): Safe Deep Model-Based RL for Sparse Cost Robotic Tasks","date":"2019-05-31","arxiv_id":"1905.13402","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extending-deep-model-predictive-control-with#ran","syntology_url":"https://syntology.ai/paper/1905.13402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13402"}},"official":null}},{"url":"/paper/sequence-modeling-of-temporal-credit","slug":"sequence-modeling-of-temporal-credit","title":"Sequence Modeling of Temporal Credit Assignment for Episodic Reinforcement Learning","date":"2019-05-31","arxiv_id":"1905.13420","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequence-modeling-of-temporal-credit#ran","syntology_url":"https://syntology.ai/paper/1905.13420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13420"}},"official":null}},{"url":"/paper/uncertainty-based-continual-learning-with","slug":"uncertainty-based-continual-learning-with","title":"Uncertainty-based Continual Learning with Adaptive Regularization","date":"2019-05-28","arxiv_id":"1905.11614","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncertainty-based-continual-learning-with#ran","syntology_url":"https://syntology.ai/paper/1905.11614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11614"}},"official":{"repos":["csm9493/UCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-np-hard-problems-on-graphs-by","slug":"solving-np-hard-problems-on-graphs-by","title":"Solving NP-Hard Problems on Graphs with Extended AlphaGo Zero","date":"2019-05-28","arxiv_id":"1905.11623","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/solving-np-hard-problems-on-graphs-by#ran","syntology_url":"https://syntology.ai/paper/1905.11623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11623"}},"official":null}},{"url":"/paper/coordinated-exploration-via-intrinsic-rewards","slug":"coordinated-exploration-via-intrinsic-rewards","title":"Coordinated Exploration via Intrinsic Rewards for Multi-Agent Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.12127","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coordinated-exploration-via-intrinsic-rewards#ran","syntology_url":"https://syntology.ai/paper/1905.12127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.12127"}},"official":{"repos":["shariqiqbal2810/Multi-Explore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finite-time-analysis-of-q-learning-with","slug":"finite-time-analysis-of-q-learning-with","title":"Finite-Sample Analysis of Nonlinear Stochastic Approximation with Applications in Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11425","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finite-time-analysis-of-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/1905.11425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11425"}},"official":null}},{"url":"/paper/tight-regret-bounds-for-model-based","slug":"tight-regret-bounds-for-model-based","title":"Tight Regret Bounds for Model-Based Reinforcement Learning with Greedy Policies","date":"2019-05-27","arxiv_id":"1905.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tight-regret-bounds-for-model-based#ran","syntology_url":"https://syntology.ai/paper/1905.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11527"}},"official":{"repos":["NMerlis/TabulaRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-policies-attacking-deep","slug":"adversarial-policies-attacking-deep","title":"Adversarial Policies: Attacking Deep Reinforcement Learning","date":"2019-05-25","arxiv_id":"1905.10615","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adversarial-policies-attacking-deep#ran","syntology_url":"https://syntology.ai/paper/1905.10615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.10615"}},"official":{"repos":["HumanCompatibleAI/adversarial-policies"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/estimating-risk-and-uncertainty-in-deep","slug":"estimating-risk-and-uncertainty-in-deep","title":"Estimating Risk and Uncertainty in Deep Reinforcement Learning","date":"2019-05-23","arxiv_id":"1905.09638","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/estimating-risk-and-uncertainty-in-deep#ran","syntology_url":"https://syntology.ai/paper/1905.09638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09638"}},"official":{"repos":["IndustAI/risk-and-uncertainty"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cobra-data-efficient-model-based-rl-through","slug":"cobra-data-efficient-model-based-rl-through","title":"COBRA: Data-Efficient Model-Based RL through Unsupervised Object Discovery and Curiosity-Driven Exploration","date":"2019-05-22","arxiv_id":"1905.09275","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cobra-data-efficient-model-based-rl-through#ran","syntology_url":"https://syntology.ai/paper/1905.09275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09275"}},"official":{"repos":["deepmind/spriteworld"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/random-expert-distillation-imitation-learning","slug":"random-expert-distillation-imitation-learning","title":"Random Expert Distillation: Imitation Learning via Expert Policy Support Estimation","date":"2019-05-16","arxiv_id":"1905.06750","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/random-expert-distillation-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1905.06750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.06750"}},"official":{"repos":["RuohanW/RED"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/expressive-priors-in-bayesian-neural-networks","slug":"expressive-priors-in-bayesian-neural-networks","title":"Expressive Priors in Bayesian Neural Networks: Kernel Combinations and Periodic Functions","date":"2019-05-15","arxiv_id":"1905.06076","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/expressive-priors-in-bayesian-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1905.06076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.06076"}},"official":null}},{"url":"/paper/control-regularization-for-reduced-variance","slug":"control-regularization-for-reduced-variance","title":"Control Regularization for Reduced Variance Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05380","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/control-regularization-for-reduced-variance#ran","syntology_url":"https://syntology.ai/paper/1905.05380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05380"}},"official":{"repos":["rcheng805/CORE-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-dynamics-priors-for-deep","slug":"task-agnostic-dynamics-priors-for-deep","title":"Task-Agnostic Dynamics Priors for Deep Reinforcement Learning","date":"2019-05-13","arxiv_id":"1905.04819","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-agnostic-dynamics-priors-for-deep#ran","syntology_url":"https://syntology.ai/paper/1905.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04819"}},"official":{"repos":["yilundu/task_agnostic_dynamics_prior"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cityflow-a-multi-agent-reinforcement-learning","slug":"cityflow-a-multi-agent-reinforcement-learning","title":"CityFlow: A Multi-Agent Reinforcement Learning Environment for Large Scale City Traffic Scenario","date":"2019-05-13","arxiv_id":"1905.05217","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cityflow-a-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1905.05217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05217"}},"official":{"repos":["cityflow-project/CityFlow"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-phase-competition-for-traffic-signal","slug":"learning-phase-competition-for-traffic-signal","title":"Learning Phase Competition for Traffic Signal Control","date":"2019-05-12","arxiv_id":"1905.04722","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-phase-competition-for-traffic-signal#ran","syntology_url":"https://syntology.ai/paper/1905.04722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04722"}},"official":null}},{"url":"/paper/colight-learning-network-level-cooperation","slug":"colight-learning-network-level-cooperation","title":"CoLight: Learning Network-level Cooperation for Traffic Signal Control","date":"2019-05-11","arxiv_id":"1905.05717","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/colight-learning-network-level-cooperation#ran","syntology_url":"https://syntology.ai/paper/1905.05717","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05717"}},"official":{"repos":["wingsweihua/colight"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dimension-wise-importance-sampling-weight","slug":"dimension-wise-importance-sampling-weight","title":"Dimension-Wise Importance Sampling Weight Clipping for Sample-Efficient Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.02363","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":12,"n_pointer_only":18,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dimension-wise-importance-sampling-weight#ran","syntology_url":"https://syntology.ai/paper/1905.02363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.02363"}},"official":{"repos":["seungyulhan/disc"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/knowing-the-what-but-not-the-where-in","slug":"knowing-the-what-but-not-the-where-in","title":"Knowing The What But Not The Where in Bayesian Optimization","date":"2019-05-07","arxiv_id":"1905.02685","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowing-the-what-but-not-the-where-in#ran","syntology_url":"https://syntology.ai/paper/1905.02685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.02685"}},"official":{"repos":["ntienvu/KnownOptimum_BO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-control-in-metric-space-with","slug":"learning-to-control-in-metric-space-with","title":"Learning to Control in Metric Space with Optimal Regret","date":"2019-05-05","arxiv_id":"1905.01576","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-in-metric-space-with#ran","syntology_url":"https://syntology.ai/paper/1905.01576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.01576"}},"official":null}},{"url":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1905.00976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00976"}},"official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rl-gan-net-a-reinforcement-learning-agent","slug":"rl-gan-net-a-reinforcement-learning-agent","title":"RL-GAN-Net: A Reinforcement Learning Agent Controlled GAN Network for Real-Time Point Cloud Shape Completion","date":"2019-04-28","arxiv_id":"1904.12304","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rl-gan-net-a-reinforcement-learning-agent#ran","syntology_url":"https://syntology.ai/paper/1904.12304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.12304"}},"official":null}}],"record_sha256":"392730f28cc23697d4fc7ce56af59f7e2635801dad491b1ec11d75185da631e2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}