{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/ran/10","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":10,"pages_in_order":12,"rows_per_page":100,"rows":[901,1000],"of":1165,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2/papers/ran/1","prev":"/task/reinforcement-learning-2/papers/ran/9","next":"/task/reinforcement-learning-2/papers/ran/11","papers":[{"url":"/paper/towards-practical-multi-object-manipulation","slug":"towards-practical-multi-object-manipulation","title":"Towards Practical Multi-Object Manipulation using Relational Reinforcement Learning","date":"2019-12-23","arxiv_id":"1912.11032","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-practical-multi-object-manipulation#ran","syntology_url":"https://syntology.ai/paper/1912.11032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11032"}},"official":null}},{"url":"/paper/interestingness-elements-for-explainable","slug":"interestingness-elements-for-explainable","title":"Interestingness Elements for Explainable Reinforcement Learning: Understanding Agents' Capabilities and Limitations","date":"2019-12-19","arxiv_id":"1912.09007","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interestingness-elements-for-explainable#ran","syntology_url":"https://syntology.ai/paper/1912.09007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.09007"}},"official":{"repos":["SRI-AIC/InterestingnessXRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distributional-reinforcement-learning-for-1","slug":"distributional-reinforcement-learning-for-1","title":"Distributional Reinforcement Learning for Energy-Based Sequential Models","date":"2019-12-18","arxiv_id":"1912.08517","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/distributional-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/1912.08517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.08517"}},"official":{"repos":["parshakova/GAMS-for-Data-Efficient-Learning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dota-2-with-large-scale-deep-reinforcement","slug":"dota-2-with-large-scale-deep-reinforcement","title":"Dota 2 with Large Scale Deep Reinforcement Learning","date":"2019-12-13","arxiv_id":"1912.06680","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dota-2-with-large-scale-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1912.06680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.06680"}},"official":null}},{"url":"/paper/measuring-the-reliability-of-reinforcement-1","slug":"measuring-the-reliability-of-reinforcement-1","title":"Measuring the Reliability of Reinforcement Learning Algorithms","date":"2019-12-10","arxiv_id":"1912.05663","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-the-reliability-of-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1912.05663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05663"}},"official":{"repos":["google-research/rl-reliability-metrics"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chainerrl-a-deep-reinforcement-learning","slug":"chainerrl-a-deep-reinforcement-learning","title":"ChainerRL: A Deep Reinforcement Learning Library","date":"2019-12-09","arxiv_id":"1912.03905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chainerrl-a-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1912.03905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03905"}},"official":{"repos":["chainer/chainerrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-object-detection-in-large-images","slug":"efficient-object-detection-in-large-images","title":"Efficient Object Detection in Large Images using Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.03966","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-object-detection-in-large-images#ran","syntology_url":"https://syntology.ai/paper/1912.03966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03966"}},"official":{"repos":["uzkent/EfficientObjectDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/safelife-10-exploring-side-effects-in-complex","slug":"safelife-10-exploring-side-effects-in-complex","title":"SafeLife 1.0: Exploring Side Effects in Complex Environments","date":"2019-12-03","arxiv_id":"1912.01217","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safelife-10-exploring-side-effects-in-complex#ran","syntology_url":"https://syntology.ai/paper/1912.01217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01217"}},"official":{"repos":["PartnershipOnAI/safelife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-procedural-generation-to-benchmark","slug":"leveraging-procedural-generation-to-benchmark","title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01588","repositories_listed":6,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leveraging-procedural-generation-to-benchmark#ran","syntology_url":"https://syntology.ai/paper/1912.01588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01588"}},"official":{"repos":["openai/procgen","openai/train-procgen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dream-to-control-learning-behaviors-by-latent","slug":"dream-to-control-learning-behaviors-by-latent","title":"Dream to Control: Learning Behaviors by Latent Imagination","date":"2019-12-03","arxiv_id":"1912.01603","repositories_listed":21,"syntology":{"n":62,"n_ran":52,"n_constructed":33,"n_ran_checked":45,"n_instrument":7,"n_unverified":10,"n_honours":1,"n_violates":0,"n_no_contract":44,"n_pointer_only":9,"phrase":"52 ran (of which 33 constructed an object rather than computing a result; 45 with no instrument failure: 1 honoured, 0 violated, 44 with no contract checked; 7 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/dream-to-control-learning-behaviors-by-latent#ran","syntology_url":"https://syntology.ai/paper/1912.01603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01603"}},"official":{"repos":["danijar/dreamer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/playing-games-in-the-dark-an-approach-for","slug":"playing-games-in-the-dark-an-approach-for","title":"Playing Games in the Dark: An approach for cross-modality transfer in reinforcement learning","date":"2019-11-28","arxiv_id":"1911.12851","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-games-in-the-dark-an-approach-for#ran","syntology_url":"https://syntology.ai/paper/1911.12851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.12851"}},"official":null}},{"url":"/paper/end-to-end-model-free-reinforcement-learning","slug":"end-to-end-model-free-reinforcement-learning","title":"End-to-End Model-Free Reinforcement Learning for Urban Driving using Implicit Affordances","date":"2019-11-25","arxiv_id":"1911.10868","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-model-free-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.10868"}},"official":{"repos":["valeoai/LearningByCheating"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-with-goal-conditioned-policies-1","slug":"planning-with-goal-conditioned-policies-1","title":"Planning with Goal-Conditioned Policies","date":"2019-11-19","arxiv_id":"1911.08453","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planning-with-goal-conditioned-policies-1#ran","syntology_url":"https://syntology.ai/paper/1911.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.08453"}},"official":null}},{"url":"/paper/combinatorial-optimization-by-graph-pointer","slug":"combinatorial-optimization-by-graph-pointer","title":"Combinatorial Optimization by Graph Pointer Networks and Hierarchical Reinforcement Learning","date":"2019-11-12","arxiv_id":"1911.04936","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/combinatorial-optimization-by-graph-pointer#ran","syntology_url":"https://syntology.ai/paper/1911.04936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04936"}},"official":{"repos":["qiang-ma/graph-pointer-network"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/drills-deep-reinforcement-learning-for-logic","slug":"drills-deep-reinforcement-learning-for-logic","title":"DRiLLS: Deep Reinforcement Learning for Logic Synthesis","date":"2019-11-11","arxiv_id":"1911.04021","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/drills-deep-reinforcement-learning-for-logic#ran","syntology_url":"https://syntology.ai/paper/1911.04021","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04021"}},"official":{"repos":["scale-lab/DRiLLS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-connected-autonomous-driving","slug":"multi-agent-connected-autonomous-driving","title":"Multi-Agent Connected Autonomous Driving using Deep Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04175","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-connected-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/1911.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04175"}},"official":{"repos":["praveen-palanisamy/macad-gym"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/real-time-reinforcement-learning","slug":"real-time-reinforcement-learning","title":"Real-Time Reinforcement Learning","date":"2019-11-11","arxiv_id":"1911.04448","repositories_listed":3,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1911.04448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04448"}},"official":{"repos":["rmst/rtrl","elementai/avenue"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-model-based-reinforcement-learning-with","slug":"a-model-based-reinforcement-learning-with","title":"Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-11-10","arxiv_id":"1911.03845","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-model-based-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1911.03845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.03845"}},"official":{"repos":["JianGuanTHU/IRecGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fully-parameterized-quantile-function-for","slug":"fully-parameterized-quantile-function-for","title":"Fully Parameterized Quantile Function for Distributional Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.02140","repositories_listed":6,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fully-parameterized-quantile-function-for#ran","syntology_url":"https://syntology.ai/paper/1911.02140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02140"}},"official":null}},{"url":"/paper/multimodal-model-agnostic-meta-learning-via","slug":"multimodal-model-agnostic-meta-learning-via","title":"Multimodal Model-Agnostic Meta-Learning via Task-Aware Modulation","date":"2019-10-30","arxiv_id":"1910.13616","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-model-agnostic-meta-learning-via#ran","syntology_url":"https://syntology.ai/paper/1910.13616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13616"}},"official":{"repos":["vuoristo/MMAML"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-to-predict-without-looking-ahead","slug":"learning-to-predict-without-looking-ahead","title":"Learning to Predict Without Looking Ahead: World Models Without Forward Prediction","date":"2019-10-29","arxiv_id":"1910.13038","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-predict-without-looking-ahead#ran","syntology_url":"https://syntology.ai/paper/1910.13038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13038"}},"official":null}},{"url":"/paper/entity-abstraction-in-visual-model-based","slug":"entity-abstraction-in-visual-model-based","title":"Entity Abstraction in Visual Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/entity-abstraction-in-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/1910.12827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12827"}},"official":{"repos":["jcoreyes/OP3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bail-best-action-imitation-learning-for-batch-1","slug":"bail-best-action-imitation-learning-for-batch-1","title":"BAIL: Best-Action Imitation Learning for Batch Deep Reinforcement Learning","date":"2019-10-27","arxiv_id":"1910.12179","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bail-best-action-imitation-learning-for-batch-1#ran","syntology_url":"https://syntology.ai/paper/1910.12179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12179"}},"official":{"repos":["lanyavik/BAIL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zpd-teaching-strategies-for-deep","slug":"zpd-teaching-strategies-for-deep","title":"ZPD Teaching Strategies for Deep Reinforcement Learning from Demonstrations","date":"2019-10-26","arxiv_id":"1910.12154","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zpd-teaching-strategies-for-deep#ran","syntology_url":"https://syntology.ai/paper/1910.12154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12154"}},"official":null}},{"url":"/paper/meta-world-a-benchmark-and-evaluation-for","slug":"meta-world-a-benchmark-and-evaluation-for","title":"Meta-World: A Benchmark and Evaluation for Multi-Task and Meta Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.10897","repositories_listed":9,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meta-world-a-benchmark-and-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/1910.10897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10897"}},"official":{"repos":["rlworkgroup/metaworld"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/hrl4in-hierarchical-reinforcement-learning","slug":"hrl4in-hierarchical-reinforcement-learning","title":"HRL4IN: Hierarchical Reinforcement Learning for Interactive Navigation with Mobile Manipulators","date":"2019-10-24","arxiv_id":"1910.11432","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hrl4in-hierarchical-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.11432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11432"}},"official":null}},{"url":"/paper/robust-domain-randomization-for-reinforcement","slug":"robust-domain-randomization-for-reinforcement","title":"Robust Visual Domain Randomization for Reinforcement Learning","date":"2019-10-23","arxiv_id":"1910.10537","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-domain-randomization-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1910.10537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10537"}},"official":{"repos":["IndustAI/visual-domain-randomization","uncharted-technologies/robust-domain-randomization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-imagined-goals-for-self-supervised","slug":"contextual-imagined-goals-for-self-supervised","title":"Contextual Imagined Goals for Self-Supervised Robotic Learning","date":"2019-10-23","arxiv_id":"1910.11670","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contextual-imagined-goals-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/1910.11670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11670"}},"official":null}},{"url":"/paper/learning-to-map-natural-language-instructions","slug":"learning-to-map-natural-language-instructions","title":"Learning to Map Natural Language Instructions to Physical Quadcopter Control using Simulated Flight","date":"2019-10-21","arxiv_id":"1910.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-map-natural-language-instructions#ran","syntology_url":"https://syntology.ai/paper/1910.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09664"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/soft-actor-critic-for-discrete-action","slug":"soft-actor-critic-for-discrete-action","title":"Soft Actor-Critic for Discrete Action Settings","date":"2019-10-16","arxiv_id":"1910.07207","repositories_listed":13,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":6,"n_honours":2,"n_violates":1,"n_no_contract":9,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/soft-actor-critic-for-discrete-action#ran","syntology_url":"https://syntology.ai/paper/1910.07207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07207"}},"official":{"repos":["p-christ/Deep-Reinforcement-Learning-Algorithms-with-PyTorch"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformers-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1910.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06764"}},"official":null}},{"url":"/paper/influence-based-multi-agent-exploration","slug":"influence-based-multi-agent-exploration","title":"Influence-Based Multi-Agent Exploration","date":"2019-10-12","arxiv_id":"1910.05512","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/influence-based-multi-agent-exploration#ran","syntology_url":"https://syntology.ai/paper/1910.05512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05512"}},"official":{"repos":["TonghanWang/EITI-EDTI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-randomization-technique-for-1","slug":"a-simple-randomization-technique-for-1","title":"Network Randomization: A Simple Technique for Generalization in Deep Reinforcement Learning","date":"2019-10-11","arxiv_id":"1910.05396","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-randomization-technique-for-1#ran","syntology_url":"https://syntology.ai/paper/1910.05396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05396"}},"official":{"repos":["pokaxpoka/netrand"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlcard-a-toolkit-for-reinforcement-learning","slug":"rlcard-a-toolkit-for-reinforcement-learning","title":"RLCard: A Toolkit for Reinforcement Learning in Card Games","date":"2019-10-10","arxiv_id":"1910.04376","repositories_listed":9,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rlcard-a-toolkit-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.04376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04376"}},"official":{"repos":["datamllab/rlcard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-paced-contextual-reinforcement-learning","slug":"self-paced-contextual-reinforcement-learning","title":"Self-Paced Contextual Reinforcement Learning","date":"2019-10-07","arxiv_id":"1910.02826","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-paced-contextual-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1910.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.02826"}},"official":{"repos":["psclklnk/self-paced-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantized-reinforcement-learning-quarl","slug":"quantized-reinforcement-learning-quarl","title":"QuaRL: Quantization for Fast and Environmentally Sustainable Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantized-reinforcement-learning-quarl#ran","syntology_url":"https://syntology.ai/paper/1910.01055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01055"}},"official":{"repos":["harvard-edge/quarl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-sample-efficiency-in-model-free-1","slug":"improving-sample-efficiency-in-model-free-1","title":"Improving Sample Efficiency in Model-Free Reinforcement Learning from Images","date":"2019-10-02","arxiv_id":"1910.01741","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-sample-efficiency-in-model-free-1#ran","syntology_url":"https://syntology.ai/paper/1910.01741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.01741"}},"official":{"repos":["denisyarats/pytorch_sac_ae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/harnessing-structures-for-value-based","slug":"harnessing-structures-for-value-based","title":"Harnessing Structures for Value-Based Planning and Reinforcement Learning","date":"2019-09-26","arxiv_id":"1909.12255","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-structures-for-value-based#ran","syntology_url":"https://syntology.ai/paper/1909.12255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.12255"}},"official":{"repos":["YyzHarry/SV-RL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/data-valuation-using-reinforcement-learning","slug":"data-valuation-using-reinforcement-learning","title":"Data Valuation using Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11671","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/data-valuation-using-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1909.11671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11671"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/emergent-tool-use-from-multi-agent","slug":"emergent-tool-use-from-multi-agent","title":"Emergent Tool Use From Multi-Agent Autocurricula","date":"2019-09-17","arxiv_id":"1909.07528","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergent-tool-use-from-multi-agent#ran","syntology_url":"https://syntology.ai/paper/1909.07528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07528"}},"official":{"repos":["openai/multi-agent-emergence-environments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mdp-playground-meta-features-in-reinforcement","slug":"mdp-playground-meta-features-in-reinforcement","title":"MDP Playground: An Analysis and Debug Testbed for Reinforcement Learning","date":"2019-09-17","arxiv_id":"1909.07750","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-playground-meta-features-in-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1909.07750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07750"}},"official":{"repos":["automl/mdp-playground"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recsim-a-configurable-simulation-platform-for","slug":"recsim-a-configurable-simulation-platform-for","title":"RecSim: A Configurable Simulation Platform for Recommender Systems","date":"2019-09-11","arxiv_id":"1909.04847","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recsim-a-configurable-simulation-platform-for#ran","syntology_url":"https://syntology.ai/paper/1909.04847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04847"}},"official":{"repos":["google-research/recsim"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-temporal-logic","slug":"reinforcement-learning-for-temporal-logic","title":"Reinforcement Learning for Temporal Logic Control Synthesis with Probabilistic Satisfaction Guarantees","date":"2019-09-11","arxiv_id":"1909.05304","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-for-temporal-logic#ran","syntology_url":"https://syntology.ai/paper/1909.05304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05304"}},"official":{"repos":["grockious/lcrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-reproducibility-by-evaluating","slug":"a-survey-on-reproducibility-by-evaluating","title":"A Survey on Reproducibility by Evaluating Deep Reinforcement Learning Algorithms on Real-World Robots","date":"2019-09-09","arxiv_id":"1909.03772","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-survey-on-reproducibility-by-evaluating#ran","syntology_url":"https://syntology.ai/paper/1909.03772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.03772"}},"official":{"repos":["dti-research/SenseActExperiments"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploratory-combinatorial-optimization-with","slug":"exploratory-combinatorial-optimization-with","title":"Exploratory Combinatorial Optimization with Reinforcement Learning","date":"2019-09-09","arxiv_id":"1909.04063","repositories_listed":2,"syntology":{"n":20,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/exploratory-combinatorial-optimization-with#ran","syntology_url":"https://syntology.ai/paper/1909.04063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.04063"}},"official":{"repos":["tomdbar/eco-dqn"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/classification-with-costly-features-as-a","slug":"classification-with-costly-features-as-a","title":"Classification with Costly Features as a Sequential Decision-Making Problem","date":"2019-09-05","arxiv_id":"1909.02564","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/classification-with-costly-features-as-a#ran","syntology_url":"https://syntology.ai/paper/1909.02564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02564"}},"official":{"repos":["jaromiru/cwcf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/answers-unite-unsupervised-metrics-for","slug":"answers-unite-unsupervised-metrics-for","title":"Answers Unite! Unsupervised Metrics for Reinforced Summarization Models","date":"2019-09-04","arxiv_id":"1909.01610","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answers-unite-unsupervised-metrics-for#ran","syntology_url":"https://syntology.ai/paper/1909.01610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01610"}},"official":{"repos":["recitalAI/summa-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlpyt-a-research-code-base-for-deep","slug":"rlpyt-a-research-code-base-for-deep","title":"rlpyt: A Research Code Base for Deep Reinforcement Learning in PyTorch","date":"2019-09-03","arxiv_id":"1909.01500","repositories_listed":9,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rlpyt-a-research-code-base-for-deep#ran","syntology_url":"https://syntology.ai/paper/1909.01500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01500"}},"official":{"repos":["astooke/rlpyt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/an-empirical-comparison-on-imitation-learning","slug":"an-empirical-comparison-on-imitation-learning","title":"An Empirical Comparison on Imitation Learning and Reinforcement Learning for Paraphrase Generation","date":"2019-08-28","arxiv_id":"1908.10835","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-empirical-comparison-on-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1908.10835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.10835"}},"official":{"repos":["ddddwy/Reinforce-Paraphrase-Generation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/openspiel-a-framework-for-reinforcement","slug":"openspiel-a-framework-for-reinforcement","title":"OpenSpiel: A Framework for Reinforcement Learning in Games","date":"2019-08-26","arxiv_id":"1908.09453","repositories_listed":16,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openspiel-a-framework-for-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1908.09453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.09453"}},"official":{"repos":["deepmind/open_spiel"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/a-generalized-algorithm-for-multi-objective","slug":"a-generalized-algorithm-for-multi-objective","title":"A Generalized Algorithm for Multi-Objective Reinforcement Learning and Policy Adaptation","date":"2019-08-21","arxiv_id":"1908.08342","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-generalized-algorithm-for-multi-objective#ran","syntology_url":"https://syntology.ai/paper/1908.08342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.08342"}},"official":{"repos":["RunzheYang/MORL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-in-world-earth","slug":"deep-reinforcement-learning-in-world-earth","title":"Deep reinforcement learning in World-Earth system models to discover sustainable management strategies","date":"2019-08-15","arxiv_id":"1908.05567","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deep-reinforcement-learning-in-world-earth#ran","syntology_url":"https://syntology.ai/paper/1908.05567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.05567"}},"official":{"repos":["fstrnad/pyDRLinWESM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-graph-to","slug":"reinforcement-learning-based-graph-to","title":"Reinforcement Learning Based Graph-to-Sequence Model for Natural Question Generation","date":"2019-08-14","arxiv_id":"1908.04942","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-based-graph-to#ran","syntology_url":"https://syntology.ai/paper/1908.04942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.04942"}},"official":{"repos":["hugochan/RL-based-Graph2Seq-for-NQG"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/behaviour-suite-for-reinforcement-learning","slug":"behaviour-suite-for-reinforcement-learning","title":"Behaviour Suite for Reinforcement Learning","date":"2019-08-09","arxiv_id":"1908.03568","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/behaviour-suite-for-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1908.03568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.03568"}},"official":{"repos":["deepmind/bsuite"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/dueling-posterior-sampling-for-preference","slug":"dueling-posterior-sampling-for-preference","title":"Dueling Posterior Sampling for Preference-Based Reinforcement Learning","date":"2019-08-04","arxiv_id":"1908.01289","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dueling-posterior-sampling-for-preference#ran","syntology_url":"https://syntology.ai/paper/1908.01289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.01289"}},"official":{"repos":["ernovoseller/DuelingPosteriorSampling"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-learning-for-efficient-reinforcement","slug":"reward-learning-for-efficient-reinforcement","title":"Reward Learning for Efficient Reinforcement Learning in Extractive Document Summarisation","date":"2019-07-30","arxiv_id":"1907.12894","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-learning-for-efficient-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.12894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.12894"}},"official":{"repos":["UKPLab/ijcai2019-relis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/google-research-football-a-novel","slug":"google-research-football-a-novel","title":"Google Research Football: A Novel Reinforcement Learning Environment","date":"2019-07-25","arxiv_id":"1907.11180","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/google-research-football-a-novel#ran","syntology_url":"https://syntology.ai/paper/1907.11180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.11180"}},"official":{"repos":["google-research/football"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/characterizing-attacks-on-deep-reinforcement","slug":"characterizing-attacks-on-deep-reinforcement","title":"Characterizing Attacks on Deep Reinforcement Learning","date":"2019-07-21","arxiv_id":"1907.09470","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/characterizing-attacks-on-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.09470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.09470"}},"official":null}},{"url":"/paper/gpu-accelerated-atari-emulation-for","slug":"gpu-accelerated-atari-emulation-for","title":"Accelerating Reinforcement Learning through GPU Atari Emulation","date":"2019-07-19","arxiv_id":"1907.08467","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpu-accelerated-atari-emulation-for#ran","syntology_url":"https://syntology.ai/paper/1907.08467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.08467"}},"official":{"repos":["NVLABs/cule"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/attentive-multi-task-deep-reinforcement","slug":"attentive-multi-task-deep-reinforcement","title":"Attentive Multi-Task Deep Reinforcement Learning","date":"2019-07-05","arxiv_id":"1907.02874","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/attentive-multi-task-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.02874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.02874"}},"official":{"repos":["braemt/attentive-multi-task-deep-reinforcement-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/conservative-q-improvement-reinforcement","slug":"conservative-q-improvement-reinforcement","title":"Conservative Q-Improvement: Reinforcement Learning for an Interpretable Decision-Tree Policy","date":"2019-07-02","arxiv_id":"1907.01180","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conservative-q-improvement-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.01180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01180"}},"official":null}},{"url":"/paper/stochastic-latent-actor-critic-deep","slug":"stochastic-latent-actor-critic-deep","title":"Stochastic Latent Actor-Critic: Deep Reinforcement Learning with a Latent Variable Model","date":"2019-07-01","arxiv_id":"1907.00953","repositories_listed":9,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stochastic-latent-actor-critic-deep#ran","syntology_url":"https://syntology.ai/paper/1907.00953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.00953"}},"official":null}},{"url":"/paper/way-off-policy-batch-deep-reinforcement","slug":"way-off-policy-batch-deep-reinforcement","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","date":"2019-06-30","arxiv_id":"1907.00456","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/way-off-policy-batch-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1907.00456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.00456"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyp-rl-hyperparameter-optimization-by","slug":"hyp-rl-hyperparameter-optimization-by","title":"Hyp-RL : Hyperparameter Optimization by Reinforcement Learning","date":"2019-06-27","arxiv_id":"1906.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyp-rl-hyperparameter-optimization-by#ran","syntology_url":"https://syntology.ai/paper/1906.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11527"}},"official":{"repos":["hadijomaa/HypRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proximal-distilled-evolutionary-reinforcement","slug":"proximal-distilled-evolutionary-reinforcement","title":"Proximal Distilled Evolutionary Reinforcement Learning","date":"2019-06-24","arxiv_id":"1906.09807","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proximal-distilled-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.09807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09807"}},"official":{"repos":["crisbodnar/pderl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-convex","slug":"reinforcement-learning-with-convex","title":"Reinforcement Learning with Convex Constraints","date":"2019-06-21","arxiv_id":"1906.09323","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-convex#ran","syntology_url":"https://syntology.ai/paper/1906.09323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.09323"}},"official":{"repos":["xkianteb/ApproPO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-trust-your-model-model-based-policy","slug":"when-to-trust-your-model-model-based-policy","title":"When to Trust Your Model: Model-Based Policy Optimization","date":"2019-06-19","arxiv_id":"1906.08253","repositories_listed":11,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-to-trust-your-model-model-based-policy#ran","syntology_url":"https://syntology.ai/paper/1906.08253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.08253"}},"official":{"repos":["JannerM/mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/unsupervised-learning-of-object-keypoints-for","slug":"unsupervised-learning-of-object-keypoints-for","title":"Unsupervised Learning of Object Keypoints for Perception and Control","date":"2019-06-19","arxiv_id":"1906.11883","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-learning-of-object-keypoints-for#ran","syntology_url":"https://syntology.ai/paper/1906.11883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11883"}},"official":{"repos":["deepmind/deepmind-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/language-as-an-abstraction-for-hierarchical","slug":"language-as-an-abstraction-for-hierarchical","title":"Language as an Abstraction for Hierarchical Deep Reinforcement Learning","date":"2019-06-18","arxiv_id":"1906.07343","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-as-an-abstraction-for-hierarchical#ran","syntology_url":"https://syntology.ai/paper/1906.07343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.07343"}},"official":{"repos":["google-research/clevr_robot_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-soft-actor-critic-emphasizing-recent#ran","syntology_url":"https://syntology.ai/paper/1906.04009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04009"}},"official":null}},{"url":"/paper/neural-keyphrase-generation-via-reinforcement","slug":"neural-keyphrase-generation-via-reinforcement","title":"Neural Keyphrase Generation via Reinforcement Learning with Adaptive Rewards","date":"2019-06-10","arxiv_id":"1906.04106","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-keyphrase-generation-via-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1906.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04106"}},"official":{"repos":["kenchan0226/keyphrase-generation-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-low-complexity","slug":"reinforcement-learning-with-low-complexity","title":"Reinforcement Learning with Low-Complexity Liquid State Machines","date":"2019-06-04","arxiv_id":"1906.01695","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-with-low-complexity#ran","syntology_url":"https://syntology.ai/paper/1906.01695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.01695"}},"official":{"repos":["wponghiran/lsm-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/extending-deep-model-predictive-control-with","slug":"extending-deep-model-predictive-control-with","title":"Safety Augmented Value Estimation from Demonstrations (SAVED): Safe Deep Model-Based RL for Sparse Cost Robotic Tasks","date":"2019-05-31","arxiv_id":"1905.13402","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extending-deep-model-predictive-control-with#ran","syntology_url":"https://syntology.ai/paper/1905.13402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13402"}},"official":null}},{"url":"/paper/sequence-modeling-of-temporal-credit","slug":"sequence-modeling-of-temporal-credit","title":"Sequence Modeling of Temporal Credit Assignment for Episodic Reinforcement Learning","date":"2019-05-31","arxiv_id":"1905.13420","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequence-modeling-of-temporal-credit#ran","syntology_url":"https://syntology.ai/paper/1905.13420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13420"}},"official":null}},{"url":"/paper/coordinated-exploration-via-intrinsic-rewards","slug":"coordinated-exploration-via-intrinsic-rewards","title":"Coordinated Exploration via Intrinsic Rewards for Multi-Agent Reinforcement Learning","date":"2019-05-28","arxiv_id":"1905.12127","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coordinated-exploration-via-intrinsic-rewards#ran","syntology_url":"https://syntology.ai/paper/1905.12127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.12127"}},"official":{"repos":["shariqiqbal2810/Multi-Explore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finite-time-analysis-of-q-learning-with","slug":"finite-time-analysis-of-q-learning-with","title":"Finite-Sample Analysis of Nonlinear Stochastic Approximation with Applications in Reinforcement Learning","date":"2019-05-27","arxiv_id":"1905.11425","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finite-time-analysis-of-q-learning-with#ran","syntology_url":"https://syntology.ai/paper/1905.11425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11425"}},"official":null}},{"url":"/paper/tight-regret-bounds-for-model-based","slug":"tight-regret-bounds-for-model-based","title":"Tight Regret Bounds for Model-Based Reinforcement Learning with Greedy Policies","date":"2019-05-27","arxiv_id":"1905.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tight-regret-bounds-for-model-based#ran","syntology_url":"https://syntology.ai/paper/1905.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11527"}},"official":{"repos":["NMerlis/TabulaRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-policies-attacking-deep","slug":"adversarial-policies-attacking-deep","title":"Adversarial Policies: Attacking Deep Reinforcement Learning","date":"2019-05-25","arxiv_id":"1905.10615","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adversarial-policies-attacking-deep#ran","syntology_url":"https://syntology.ai/paper/1905.10615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.10615"}},"official":{"repos":["HumanCompatibleAI/adversarial-policies"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/estimating-risk-and-uncertainty-in-deep","slug":"estimating-risk-and-uncertainty-in-deep","title":"Estimating Risk and Uncertainty in Deep Reinforcement Learning","date":"2019-05-23","arxiv_id":"1905.09638","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/estimating-risk-and-uncertainty-in-deep#ran","syntology_url":"https://syntology.ai/paper/1905.09638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09638"}},"official":{"repos":["IndustAI/risk-and-uncertainty"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cobra-data-efficient-model-based-rl-through","slug":"cobra-data-efficient-model-based-rl-through","title":"COBRA: Data-Efficient Model-Based RL through Unsupervised Object Discovery and Curiosity-Driven Exploration","date":"2019-05-22","arxiv_id":"1905.09275","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cobra-data-efficient-model-based-rl-through#ran","syntology_url":"https://syntology.ai/paper/1905.09275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09275"}},"official":{"repos":["deepmind/spriteworld"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/random-expert-distillation-imitation-learning","slug":"random-expert-distillation-imitation-learning","title":"Random Expert Distillation: Imitation Learning via Expert Policy Support Estimation","date":"2019-05-16","arxiv_id":"1905.06750","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/random-expert-distillation-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/1905.06750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.06750"}},"official":{"repos":["RuohanW/RED"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/expressive-priors-in-bayesian-neural-networks","slug":"expressive-priors-in-bayesian-neural-networks","title":"Expressive Priors in Bayesian Neural Networks: Kernel Combinations and Periodic Functions","date":"2019-05-15","arxiv_id":"1905.06076","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/expressive-priors-in-bayesian-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1905.06076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.06076"}},"official":null}},{"url":"/paper/control-regularization-for-reduced-variance","slug":"control-regularization-for-reduced-variance","title":"Control Regularization for Reduced Variance Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05380","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/control-regularization-for-reduced-variance#ran","syntology_url":"https://syntology.ai/paper/1905.05380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05380"}},"official":{"repos":["rcheng805/CORE-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-dynamics-priors-for-deep","slug":"task-agnostic-dynamics-priors-for-deep","title":"Task-Agnostic Dynamics Priors for Deep Reinforcement Learning","date":"2019-05-13","arxiv_id":"1905.04819","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-agnostic-dynamics-priors-for-deep#ran","syntology_url":"https://syntology.ai/paper/1905.04819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04819"}},"official":{"repos":["yilundu/task_agnostic_dynamics_prior"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/cityflow-a-multi-agent-reinforcement-learning","slug":"cityflow-a-multi-agent-reinforcement-learning","title":"CityFlow: A Multi-Agent Reinforcement Learning Environment for Large Scale City Traffic Scenario","date":"2019-05-13","arxiv_id":"1905.05217","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cityflow-a-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1905.05217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05217"}},"official":{"repos":["cityflow-project/CityFlow"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dimension-wise-importance-sampling-weight","slug":"dimension-wise-importance-sampling-weight","title":"Dimension-Wise Importance Sampling Weight Clipping for Sample-Efficient Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.02363","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":12,"n_pointer_only":18,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dimension-wise-importance-sampling-weight#ran","syntology_url":"https://syntology.ai/paper/1905.02363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.02363"}},"official":{"repos":["seungyulhan/disc"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-control-in-metric-space-with","slug":"learning-to-control-in-metric-space-with","title":"Learning to Control in Metric Space with Optimal Regret","date":"2019-05-05","arxiv_id":"1905.01576","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-control-in-metric-space-with#ran","syntology_url":"https://syntology.ai/paper/1905.01576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.01576"}},"official":null}},{"url":"/paper/collaborative-evolutionary-reinforcement","slug":"collaborative-evolutionary-reinforcement","title":"Collaborative Evolutionary Reinforcement Learning","date":"2019-05-02","arxiv_id":"1905.00976","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-evolutionary-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1905.00976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.00976"}},"official":{"repos":["intelai/cerl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-logic-reinforcement-learning","slug":"neural-logic-reinforcement-learning","title":"Neural Logic Reinforcement Learning","date":"2019-04-24","arxiv_id":"1904.10729","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neural-logic-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1904.10729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10729"}},"official":{"repos":["ZhengyaoJiang/NLRL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/graphnas-graph-neural-architecture-search","slug":"graphnas-graph-neural-architecture-search","title":"GraphNAS: Graph Neural Architecture Search with Reinforcement Learning","date":"2019-04-22","arxiv_id":"1904.09981","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/graphnas-graph-neural-architecture-search#ran","syntology_url":"https://syntology.ai/paper/1904.09981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09981"}},"official":{"repos":["GraphNAS/GraphNAS-simple"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rogue-gym-a-new-challenge-for-generalization","slug":"rogue-gym-a-new-challenge-for-generalization","title":"Rogue-Gym: A New Challenge for Generalization in Reinforcement Learning","date":"2019-04-17","arxiv_id":"1904.08129","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rogue-gym-a-new-challenge-for-generalization#ran","syntology_url":"https://syntology.ai/paper/1904.08129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08129"}},"official":{"repos":["kngwyu/rogue-gym","kngwyu/rogue-gym-agents-cog19"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hitchhikers-guide-to-statistical","slug":"a-hitchhikers-guide-to-statistical","title":"A Hitchhiker's Guide to Statistical Comparisons of Reinforcement Learning Algorithms","date":"2019-04-15","arxiv_id":"1904.06979","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-hitchhikers-guide-to-statistical#ran","syntology_url":"https://syntology.ai/paper/1904.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06979"}},"official":{"repos":["flowersteam/rl_stats","ccolas/rl_stats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extrapolating-beyond-suboptimal","slug":"extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extrapolating-beyond-suboptimal#ran","syntology_url":"https://syntology.ai/paper/1904.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06387"}},"official":{"repos":["hiwonjoon/ICML2019-TREX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-reinforcement-learning-via","slug":"interpretable-reinforcement-learning-via","title":"Optimization Methods for Interpretable Differentiable Decision Trees in Reinforcement Learning","date":"2019-03-22","arxiv_id":"1903.09338","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interpretable-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/1903.09338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.09338"}},"official":null}},{"url":"/paper/efficient-off-policy-meta-reinforcement","slug":"efficient-off-policy-meta-reinforcement","title":"Efficient Off-Policy Meta-Reinforcement Learning via Probabilistic Context Variables","date":"2019-03-19","arxiv_id":"1903.08254","repositories_listed":7,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-off-policy-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1903.08254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.08254"}},"official":{"repos":["katerakelly/oyster"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/coacor-code-annotation-for-code-retrieval","slug":"coacor-code-annotation-for-code-retrieval","title":"CoaCor: Code Annotation for Code Retrieval with Reinforcement Learning","date":"2019-03-13","arxiv_id":"1904.00720","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/coacor-code-annotation-for-code-retrieval#ran","syntology_url":"https://syntology.ai/paper/1904.00720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.00720"}},"official":{"repos":["LittleYUYU/CoaCor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-reinforcement-learning-with-expert","slug":"hybrid-reinforcement-learning-with-expert","title":"Hybrid Reinforcement Learning with Expert State Sequences","date":"2019-03-11","arxiv_id":"1903.04110","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-reinforcement-learning-with-expert#ran","syntology_url":"https://syntology.ai/paper/1903.04110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04110"}},"official":{"repos":["XiaoxiaoGuo/tensor4rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stroke-based-artistic-rendering-agent-with","slug":"stroke-based-artistic-rendering-agent-with","title":"Learning to Paint With Model-based Deep Reinforcement Learning","date":"2019-03-11","arxiv_id":"1903.04411","repositories_listed":6,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/stroke-based-artistic-rendering-agent-with#ran","syntology_url":"https://syntology.ai/paper/1903.04411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04411"}},"official":{"repos":["hzwer/ICCV2019-LearningToPaint"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-2","slug":"multi-agent-deep-reinforcement-learning-for-2","title":"Multi-Agent Deep Reinforcement Learning for Large-scale Traffic Signal Control","date":"2019-03-11","arxiv_id":"1903.04527","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-deep-reinforcement-learning-for-2#ran","syntology_url":"https://syntology.ai/paper/1903.04527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04527"}},"official":{"repos":["cts198859/deeprl_signal_control"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-power-system-emergency-control-using","slug":"adaptive-power-system-emergency-control-using","title":"Adaptive Power System Emergency Control using Deep Reinforcement Learning","date":"2019-03-09","arxiv_id":"1903.03712","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-power-system-emergency-control-using#ran","syntology_url":"https://syntology.ai/paper/1903.03712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.03712"}},"official":{"repos":["RLGC-Project/RLGC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"f35b07631c4cded03c0205f02ff37aec66b5946c9d8e284d3c44c327d846611f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}