{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/2","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":132,"rows_per_page":100,"rows":[101,200],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning","next":"/task/reinforcement-learning/papers/3","papers":[{"url":"/paper/learning-to-drive-in-a-day","slug":"learning-to-drive-in-a-day","title":"Learning to Drive in a Day","date":"2018-07-01","arxiv_id":"1807.00412","repositories_listed":8,"syntology":{"n":22,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-to-drive-in-a-day#ran","syntology_url":"https://syntology.ai/paper/1807.00412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.00412"}},"official":null}},{"url":"/paper/accelerated-methods-for-deep-reinforcement","slug":"accelerated-methods-for-deep-reinforcement","title":"Accelerated Methods for Deep Reinforcement Learning","date":"2018-03-07","arxiv_id":"1803.02811","repositories_listed":8,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accelerated-methods-for-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1803.02811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.02811"}},"official":null}},{"url":"/paper/deepmind-control-suite","slug":"deepmind-control-suite","title":"DeepMind Control Suite","date":"2018-01-02","arxiv_id":"1801.00690","repositories_listed":8,"syntology":null},{"url":"/paper/vision-and-language-navigation-interpreting","slug":"vision-and-language-navigation-interpreting","title":"Vision-and-Language Navigation: Interpreting visually-grounded navigation instructions in real environments","date":"2017-11-20","arxiv_id":"1711.07280","repositories_listed":8,"syntology":null},{"url":"/paper/go-for-a-walk-and-arrive-at-the-answer","slug":"go-for-a-walk-and-arrive-at-the-answer","title":"Go for a Walk and Arrive at the Answer: Reasoning Over Paths in Knowledge Bases using Reinforcement Learning","date":"2017-11-15","arxiv_id":"1711.05851","repositories_listed":8,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/go-for-a-walk-and-arrive-at-the-answer#ran","syntology_url":"https://syntology.ai/paper/1711.05851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.05851"}},"official":{"repos":["shehzaadzd/MINERVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/scalable-trust-region-method-for-deep","slug":"scalable-trust-region-method-for-deep","title":"Scalable trust-region method for deep reinforcement learning using Kronecker-factored approximation","date":"2017-08-17","arxiv_id":"1708.05144","repositories_listed":8,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-trust-region-method-for-deep#ran","syntology_url":"https://syntology.ai/paper/1708.05144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1708.05144"}},"official":{"repos":["openai/baselines"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-combinatorial-optimization","slug":"learning-combinatorial-optimization","title":"Learning Combinatorial Optimization Algorithms over Graphs","date":"2017-04-05","arxiv_id":"1704.01665","repositories_listed":8,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-combinatorial-optimization#ran","syntology_url":"https://syntology.ai/paper/1704.01665","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1704.01665"}},"official":{"repos":["Hanjun-Dai/graph_comb_opt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/adversarial-learning-for-neural-dialogue","slug":"adversarial-learning-for-neural-dialogue","title":"Adversarial Learning for Neural Dialogue Generation","date":"2017-01-23","arxiv_id":"1701.06547","repositories_listed":8,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adversarial-learning-for-neural-dialogue#ran","syntology_url":"https://syntology.ai/paper/1701.06547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.06547"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-dialogue","slug":"deep-reinforcement-learning-for-dialogue","title":"Deep Reinforcement Learning for Dialogue Generation","date":"2016-06-05","arxiv_id":"1606.01541","repositories_listed":8,"syntology":null},{"url":"/paper/continuous-deep-q-learning-with-model-based","slug":"continuous-deep-q-learning-with-model-based","title":"Continuous Deep Q-Learning with Model-based Acceleration","date":"2016-03-02","arxiv_id":"1603.00748","repositories_listed":8,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continuous-deep-q-learning-with-model-based#ran","syntology_url":"https://syntology.ai/paper/1603.00748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1603.00748"}},"official":null}},{"url":"/paper/value-iteration-networks","slug":"value-iteration-networks","title":"Value Iteration Networks","date":"2016-02-09","arxiv_id":"1602.02867","repositories_listed":8,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/value-iteration-networks#ran","syntology_url":"https://syntology.ai/paper/1602.02867","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1602.02867"}},"official":{"repos":["avivt/VIN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/human-level-control-through-deep","slug":"human-level-control-through-deep","title":"Human level control through deep reinforcement learning","date":"2015-02-25","arxiv_id":null,"repositories_listed":8,"syntology":null},{"url":"/paper/mastering-diverse-domains-through-world","slug":"mastering-diverse-domains-through-world","title":"Mastering Diverse Domains through World Models","date":"2023-01-10","arxiv_id":"2301.04104","repositories_listed":7,"syntology":{"n":34,"n_ran":21,"n_constructed":0,"n_ran_checked":16,"n_instrument":5,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 5 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/mastering-diverse-domains-through-world#ran","syntology_url":"https://syntology.ai/paper/2301.04104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.04104"}},"official":{"repos":["danijar/dreamerv3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/datasets-for-data-driven-reinforcement","slug":"datasets-for-data-driven-reinforcement","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","date":"2020-04-15","arxiv_id":"2004.07219","repositories_listed":7,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/datasets-for-data-driven-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.07219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07219"}},"official":{"repos":["rail-berkeley/d4rl","rail-berkeley/offline_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/curl-contrastive-unsupervised-representations","slug":"curl-contrastive-unsupervised-representations","title":"CURL: Contrastive Unsupervised Representations for Reinforcement Learning","date":"2020-04-08","arxiv_id":"2004.04136","repositories_listed":7,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/curl-contrastive-unsupervised-representations#ran","syntology_url":"https://syntology.ai/paper/2004.04136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.04136"}},"official":{"repos":["MishaLaskin/curl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gridmask-data-augmentation","slug":"gridmask-data-augmentation","title":"GridMask Data Augmentation","date":"2020-01-13","arxiv_id":"2001.04086","repositories_listed":7,"syntology":null},{"url":"/paper/training-agents-using-upside-down","slug":"training-agents-using-upside-down","title":"Training Agents using Upside-Down Reinforcement Learning","date":"2019-12-05","arxiv_id":"1912.02877","repositories_listed":7,"syntology":null},{"url":"/paper/efficient-off-policy-meta-reinforcement","slug":"efficient-off-policy-meta-reinforcement","title":"Efficient Off-Policy Meta-Reinforcement Learning via Probabilistic Context Variables","date":"2019-03-19","arxiv_id":"1903.08254","repositories_listed":7,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-off-policy-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1903.08254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.08254"}},"official":{"repos":["katerakelly/oyster"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/optimization-of-molecules-via-deep","slug":"optimization-of-molecules-via-deep","title":"Optimization of Molecules via Deep Reinforcement Learning","date":"2018-10-19","arxiv_id":"1810.08678","repositories_listed":7,"syntology":null},{"url":"/paper/crowd-robot-interaction-crowd-aware-robot","slug":"crowd-robot-interaction-crowd-aware-robot","title":"Crowd-Robot Interaction: Crowd-aware Robot Navigation with Attention-based Deep Reinforcement Learning","date":"2018-09-24","arxiv_id":"1809.08835","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crowd-robot-interaction-crowd-aware-robot#ran","syntology_url":"https://syntology.ai/paper/1809.08835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.08835"}},"official":{"repos":["vita-epfl/CrowdNav","vita-epfl/DyNav"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/vizdoom-competitions-playing-doom-from-pixels","slug":"vizdoom-competitions-playing-doom-from-pixels","title":"ViZDoom Competitions: Playing Doom from Pixels","date":"2018-09-10","arxiv_id":"1809.03470","repositories_listed":7,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/vizdoom-competitions-playing-doom-from-pixels#ran","syntology_url":"https://syntology.ai/paper/1809.03470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.03470"}},"official":{"repos":["mwydmuch/ViZDoom"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/relational-deep-reinforcement-learning","slug":"relational-deep-reinforcement-learning","title":"Relational Deep Reinforcement Learning","date":"2018-06-05","arxiv_id":"1806.01830","repositories_listed":7,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/relational-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1806.01830","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01830"}},"official":null}},{"url":"/paper/fractal-ai-a-fragile-theory-of-intelligence","slug":"fractal-ai-a-fragile-theory-of-intelligence","title":"Fractal AI: A fragile theory of intelligence","date":"2018-03-13","arxiv_id":"1803.05049","repositories_listed":7,"syntology":null},{"url":"/paper/some-considerations-on-learning-to-explore","slug":"some-considerations-on-learning-to-explore","title":"Some Considerations on Learning to Explore via Meta-Reinforcement Learning","date":"2018-03-03","arxiv_id":"1803.01118","repositories_listed":7,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-list-wise","slug":"deep-reinforcement-learning-for-list-wise","title":"Deep Reinforcement Learning for List-wise Recommendations","date":"2017-12-30","arxiv_id":"1801.00209","repositories_listed":7,"syntology":null},{"url":"/paper/backpropagation-through-the-void-optimizing","slug":"backpropagation-through-the-void-optimizing","title":"Backpropagation through the Void: Optimizing control variates for black-box gradient estimation","date":"2017-10-31","arxiv_id":"1711.00123","repositories_listed":7,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":3,"n_violates":1,"n_no_contract":3,"n_pointer_only":12,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/backpropagation-through-the-void-optimizing#ran","syntology_url":"https://syntology.ai/paper/1711.00123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.00123"}},"official":{"repos":["duvenaud/relax"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-robust-rewards-with-adversarial","slug":"learning-robust-rewards-with-adversarial","title":"Learning Robust Rewards with Adversarial Inverse Reinforcement Learning","date":"2017-10-30","arxiv_id":"1710.11248","repositories_listed":7,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-robust-rewards-with-adversarial#ran","syntology_url":"https://syntology.ai/paper/1710.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11248"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-from-human","slug":"deep-reinforcement-learning-from-human","title":"Deep reinforcement learning from human preferences","date":"2017-06-12","arxiv_id":"1706.03741","repositories_listed":7,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-from-human#ran","syntology_url":"https://syntology.ai/paper/1706.03741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1706.03741"}},"official":null}},{"url":"/paper/counterfactual-multi-agent-policy-gradients","slug":"counterfactual-multi-agent-policy-gradients","title":"Counterfactual Multi-Agent Policy Gradients","date":"2017-05-24","arxiv_id":"1705.08926","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-multi-agent-policy-gradients#ran","syntology_url":"https://syntology.ai/paper/1705.08926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1705.08926"}},"official":null}},{"url":"/paper/time-contrastive-networks-self-supervised","slug":"time-contrastive-networks-self-supervised","title":"Time-Contrastive Networks: Self-Supervised Learning from Video","date":"2017-04-23","arxiv_id":"1704.06888","repositories_listed":7,"syntology":null},{"url":"/paper/learning-cooperative-visual-dialog-agents","slug":"learning-cooperative-visual-dialog-agents","title":"Learning Cooperative Visual Dialog Agents with Deep Reinforcement Learning","date":"2017-03-20","arxiv_id":"1703.06585","repositories_listed":7,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-cooperative-visual-dialog-agents#ran","syntology_url":"https://syntology.ai/paper/1703.06585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.06585"}},"official":null}},{"url":"/paper/robust-adversarial-reinforcement-learning","slug":"robust-adversarial-reinforcement-learning","title":"Robust Adversarial Reinforcement Learning","date":"2017-03-08","arxiv_id":"1703.02702","repositories_listed":7,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/robust-adversarial-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1703.02702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.02702"}},"official":null}},{"url":"/paper/sample-efficient-actor-critic-with-experience","slug":"sample-efficient-actor-critic-with-experience","title":"Sample Efficient Actor-Critic with Experience Replay","date":"2016-11-03","arxiv_id":"1611.01224","repositories_listed":7,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/sample-efficient-actor-critic-with-experience#ran","syntology_url":"https://syntology.ai/paper/1611.01224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01224"}},"official":null}},{"url":"/paper/playing-fps-games-with-deep-reinforcement","slug":"playing-fps-games-with-deep-reinforcement","title":"Playing FPS Games with Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05521","repositories_listed":7,"syntology":null},{"url":"/paper/deep-reinforcement-learning-from-self-play-in","slug":"deep-reinforcement-learning-from-self-play-in","title":"Deep Reinforcement Learning from Self-Play in Imperfect-Information Games","date":"2016-03-03","arxiv_id":"1603.01121","repositories_listed":7,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-parameterized","slug":"deep-reinforcement-learning-in-parameterized","title":"Deep Reinforcement Learning in Parameterized Action Space","date":"2015-11-13","arxiv_id":"1511.04143","repositories_listed":7,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-in-parameterized#ran","syntology_url":"https://syntology.ai/paper/1511.04143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.04143"}},"official":{"repos":["mhauskn/dqn-hfo"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/self-timed-reinforcement-learning-using","slug":"self-timed-reinforcement-learning-using","title":"Self-timed Reinforcement Learning using Tsetlin Machine","date":"2021-09-02","arxiv_id":"2109.00846","repositories_listed":6,"syntology":null},{"url":"/paper/munchausen-reinforcement-learning","slug":"munchausen-reinforcement-learning","title":"Munchausen Reinforcement Learning","date":"2020-07-28","arxiv_id":"2007.14430","repositories_listed":6,"syntology":{"n":22,"n_ran":12,"n_constructed":11,"n_ran_checked":12,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/munchausen-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2007.14430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.14430"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/accelerating-online-reinforcement-learning","slug":"accelerating-online-reinforcement-learning","title":"AWAC: Accelerating Online Reinforcement Learning with Offline Datasets","date":"2020-06-16","arxiv_id":"2006.09359","repositories_listed":6,"syntology":{"n":13,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":8,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/accelerating-online-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2006.09359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09359"}},"official":{"repos":["vitchyr/rlkit"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/never-give-up-learning-directed-exploration-1","slug":"never-give-up-learning-directed-exploration-1","title":"Never Give Up: Learning Directed Exploration Strategies","date":"2020-02-14","arxiv_id":"2002.06038","repositories_listed":6,"syntology":{"n":14,"n_ran":7,"n_constructed":3,"n_ran_checked":5,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/never-give-up-learning-directed-exploration-1#ran","syntology_url":"https://syntology.ai/paper/2002.06038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06038"}},"official":null}},{"url":"/paper/pcgrl-procedural-content-generation-via","slug":"pcgrl-procedural-content-generation-via","title":"PCGRL: Procedural Content Generation via Reinforcement Learning","date":"2020-01-24","arxiv_id":"2001.09212","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pcgrl-procedural-content-generation-via#ran","syntology_url":"https://syntology.ai/paper/2001.09212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.09212"}},"official":{"repos":["amidos2006/gym-pcgrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/covariance-matrix-adaptation-for-the-rapid","slug":"covariance-matrix-adaptation-for-the-rapid","title":"Covariance Matrix Adaptation for the Rapid Illumination of Behavior Space","date":"2019-12-05","arxiv_id":"1912.02400","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/covariance-matrix-adaptation-for-the-rapid#ran","syntology_url":"https://syntology.ai/paper/1912.02400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.02400"}},"official":{"repos":["tehqin/EvoStone","tehqin/QualDivBenchmark"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/leveraging-procedural-generation-to-benchmark","slug":"leveraging-procedural-generation-to-benchmark","title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","date":"2019-12-03","arxiv_id":"1912.01588","repositories_listed":6,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/leveraging-procedural-generation-to-benchmark#ran","syntology_url":"https://syntology.ai/paper/1912.01588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.01588"}},"official":{"repos":["openai/procgen","openai/train-procgen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/fully-parameterized-quantile-function-for","slug":"fully-parameterized-quantile-function-for","title":"Fully Parameterized Quantile Function for Distributional Reinforcement Learning","date":"2019-11-05","arxiv_id":"1911.02140","repositories_listed":6,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fully-parameterized-quantile-function-for#ran","syntology_url":"https://syntology.ai/paper/1911.02140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.02140"}},"official":null}},{"url":"/paper/searching-for-a-robust-neural-architecture-in-1","slug":"searching-for-a-robust-neural-architecture-in-1","title":"Searching for A Robust Neural Architecture in Four GPU Hours","date":"2019-10-10","arxiv_id":"1910.04465","repositories_listed":6,"syntology":null},{"url":"/paper/advantage-weighted-regression-simple-and","slug":"advantage-weighted-regression-simple-and","title":"Advantage-Weighted Regression: Simple and Scalable Off-Policy Reinforcement Learning","date":"2019-10-01","arxiv_id":"1910.00177","repositories_listed":6,"syntology":null},{"url":"/paper/unsupervised-learning-of-object-keypoints-for","slug":"unsupervised-learning-of-object-keypoints-for","title":"Unsupervised Learning of Object Keypoints for Perception and Control","date":"2019-06-19","arxiv_id":"1906.11883","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-learning-of-object-keypoints-for#ran","syntology_url":"https://syntology.ai/paper/1906.11883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.11883"}},"official":{"repos":["deepmind/deepmind-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stroke-based-artistic-rendering-agent-with","slug":"stroke-based-artistic-rendering-agent-with","title":"Learning to Paint With Model-based Deep Reinforcement Learning","date":"2019-03-11","arxiv_id":"1903.04411","repositories_listed":6,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/stroke-based-artistic-rendering-agent-with#ran","syntology_url":"https://syntology.ai/paper/1903.04411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.04411"}},"official":{"repos":["hzwer/ICCV2019-LearningToPaint"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/nscaching-simple-and-efficient-negative","slug":"nscaching-simple-and-efficient-negative","title":"NSCaching: Simple and Efficient Negative Sampling for Knowledge Graph Embedding","date":"2018-12-16","arxiv_id":"1812.06410","repositories_listed":6,"syntology":null},{"url":"/paper/learning-to-compose-dynamic-tree-structures","slug":"learning-to-compose-dynamic-tree-structures","title":"Learning to Compose Dynamic Tree Structures for Visual Contexts","date":"2018-12-05","arxiv_id":"1812.01880","repositories_listed":6,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-to-compose-dynamic-tree-structures#ran","syntology_url":"https://syntology.ai/paper/1812.01880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.01880"}},"official":null}},{"url":"/paper/promp-proximal-meta-policy-search","slug":"promp-proximal-meta-policy-search","title":"ProMP: Proximal Meta-Policy Search","date":"2018-10-16","arxiv_id":"1810.06784","repositories_listed":6,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/promp-proximal-meta-policy-search#ran","syntology_url":"https://syntology.ai/paper/1810.06784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06784"}},"official":{"repos":["jonasrothfuss/promp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/near-optimal-representation-learning-for","slug":"near-optimal-representation-learning-for","title":"Near-Optimal Representation Learning for Hierarchical Reinforcement Learning","date":"2018-10-02","arxiv_id":"1810.01257","repositories_listed":6,"syntology":null},{"url":"/paper/evolution-guided-policy-gradient-in","slug":"evolution-guided-policy-gradient-in","title":"Evolution-Guided Policy Gradient in Reinforcement Learning","date":"2018-05-21","arxiv_id":"1805.07917","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evolution-guided-policy-gradient-in#ran","syntology_url":"https://syntology.ai/paper/1805.07917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07917"}},"official":{"repos":["ShawK91/erl_paper_nips18"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/motion-planning-among-dynamic-decision-making","slug":"motion-planning-among-dynamic-decision-making","title":"Motion Planning Among Dynamic, Decision-Making Agents with Deep Reinforcement Learning","date":"2018-05-04","arxiv_id":"1805.01956","repositories_listed":6,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/motion-planning-among-dynamic-decision-making#ran","syntology_url":"https://syntology.ai/paper/1805.01956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.01956"}},"official":{"repos":["mfe7/cadrl_ros"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deepmimic-example-guided-deep-reinforcement","slug":"deepmimic-example-guided-deep-reinforcement","title":"DeepMimic: Example-Guided Deep Reinforcement Learning of Physics-Based Character Skills","date":"2018-04-08","arxiv_id":"1804.02717","repositories_listed":6,"syntology":null},{"url":"/paper/safe-exploration-in-continuous-action-spaces","slug":"safe-exploration-in-continuous-action-spaces","title":"Safe Exploration in Continuous Action Spaces","date":"2018-01-26","arxiv_id":"1801.08757","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/safe-exploration-in-continuous-action-spaces#ran","syntology_url":"https://syntology.ai/paper/1801.08757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.08757"}},"official":null}},{"url":"/paper/deeptraffic-crowdsourced-hyperparameter","slug":"deeptraffic-crowdsourced-hyperparameter","title":"DeepTraffic: Crowdsourced Hyperparameter Tuning of Deep Reinforcement Learning Systems for Multi-Agent Dense Traffic Navigation","date":"2018-01-09","arxiv_id":"1801.02805","repositories_listed":6,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-unsupervised","slug":"deep-reinforcement-learning-for-unsupervised","title":"Deep Reinforcement Learning for Unsupervised Video Summarization with Diversity-Representativeness Reward","date":"2017-12-29","arxiv_id":"1801.00054","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/1801.00054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.00054"}},"official":{"repos":["KaiyangZhou/vsumm-reinforce"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/pixelsnail-an-improved-autoregressive","slug":"pixelsnail-an-improved-autoregressive","title":"PixelSNAIL: An Improved Autoregressive Generative Model","date":"2017-12-28","arxiv_id":"1712.09763","repositories_listed":6,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pixelsnail-an-improved-autoregressive#ran","syntology_url":"https://syntology.ai/paper/1712.09763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.09763"}},"official":{"repos":["neocxi/pixelsnail-public"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stochastic-answer-networks-for-machine","slug":"stochastic-answer-networks-for-machine","title":"Stochastic Answer Networks for Machine Reading Comprehension","date":"2017-12-10","arxiv_id":"1712.03556","repositories_listed":6,"syntology":null},{"url":"/paper/long-text-generation-via-adversarial-training","slug":"long-text-generation-via-adversarial-training","title":"Long Text Generation via Adversarial Training with Leaked Information","date":"2017-09-24","arxiv_id":"1709.08624","repositories_listed":6,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/long-text-generation-via-adversarial-training#ran","syntology_url":"https://syntology.ai/paper/1709.08624","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.08624"}},"official":{"repos":["CR-Gjx/LeakGAN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-with-opponent-learning-awareness","slug":"learning-with-opponent-learning-awareness","title":"Learning with Opponent-Learning Awareness","date":"2017-09-13","arxiv_id":"1709.04326","repositories_listed":6,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-with-opponent-learning-awareness#ran","syntology_url":"https://syntology.ai/paper/1709.04326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.04326"}},"official":{"repos":["alshedivat/lola"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/solving-high-dimensional-partial-differential","slug":"solving-high-dimensional-partial-differential","title":"Solving high-dimensional partial differential equations using deep learning","date":"2017-07-09","arxiv_id":"1707.02568","repositories_listed":6,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/solving-high-dimensional-partial-differential#ran","syntology_url":"https://syntology.ai/paper/1707.02568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.02568"}},"official":null}},{"url":"/paper/emergence-of-locomotion-behaviours-in-rich","slug":"emergence-of-locomotion-behaviours-in-rich","title":"Emergence of Locomotion Behaviours in Rich Environments","date":"2017-07-07","arxiv_id":"1707.02286","repositories_listed":6,"syntology":null},{"url":"/paper/criticality-as-it-could-be-organizational","slug":"criticality-as-it-could-be-organizational","title":"Criticality as It Could Be: organizational invariance as self-organized criticality in embodied agents","date":"2017-04-18","arxiv_id":"1704.05255","repositories_listed":6,"syntology":null},{"url":"/paper/virtual-to-real-reinforcement-learning-for","slug":"virtual-to-real-reinforcement-learning-for","title":"Virtual to Real Reinforcement Learning for Autonomous Driving","date":"2017-04-13","arxiv_id":"1704.03952","repositories_listed":6,"syntology":null},{"url":"/paper/deep-q-learning-from-demonstrations","slug":"deep-q-learning-from-demonstrations","title":"Deep Q-learning from Demonstrations","date":"2017-04-12","arxiv_id":"1704.03732","repositories_listed":6,"syntology":null},{"url":"/paper/deep-exploration-via-bootstrapped-dqn","slug":"deep-exploration-via-bootstrapped-dqn","title":"Deep Exploration via Bootstrapped DQN","date":"2016-02-15","arxiv_id":"1602.04621","repositories_listed":6,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-exploration-via-bootstrapped-dqn#ran","syntology_url":"https://syntology.ai/paper/1602.04621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1602.04621"}},"official":null}},{"url":"/paper/lima-less-is-more-for-alignment","slug":"lima-less-is-more-for-alignment","title":"LIMA: Less Is More for Alignment","date":"2023-05-18","arxiv_id":"2305.11206","repositories_listed":5,"syntology":null},{"url":"/paper/reflexion-language-agents-with-verbal","slug":"reflexion-language-agents-with-verbal","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","date":"2023-03-20","arxiv_id":"2303.11366","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reflexion-language-agents-with-verbal#ran","syntology_url":"https://syntology.ai/paper/2303.11366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11366"}},"official":{"repos":["noahshinn024/reflexion"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/corl-research-oriented-deep-offline-1","slug":"corl-research-oriented-deep-offline-1","title":"CORL: Research-oriented Deep Offline Reinforcement Learning Library","date":"2022-10-13","arxiv_id":"2210.07105","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/corl-research-oriented-deep-offline-1#ran","syntology_url":"https://syntology.ai/paper/2210.07105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07105"}},"official":{"repos":["corl-team/CORL","hanjuku-kaso/awesome-offline-rl","tinkoff-ai/CORL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-based-offline-reinforcement","slug":"uncertainty-based-offline-reinforcement","title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","date":"2021-10-04","arxiv_id":"2110.01548","repositories_listed":5,"syntology":{"n":21,"n_ran":13,"n_constructed":11,"n_ran_checked":12,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":6,"phrase":"13 ran (of which 11 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/uncertainty-based-offline-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2110.01548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01548"}},"official":null}},{"url":"/paper/safe-learning-in-robotics-from-learning-based","slug":"safe-learning-in-robotics-from-learning-based","title":"Safe Learning in Robotics: From Learning-Based Control to Safe Reinforcement Learning","date":"2021-08-13","arxiv_id":"2108.06266","repositories_listed":5,"syntology":null},{"url":"/paper/gym-m-rts-toward-affordable-full-game-real","slug":"gym-m-rts-toward-affordable-full-game-real","title":"Gym-$μ$RTS: Toward Affordable Full Game Real-time Strategy Games Research with Deep Reinforcement Learning","date":"2021-05-21","arxiv_id":"2105.13807","repositories_listed":5,"syntology":null},{"url":"/paper/evolving-reinforcement-learning-algorithms-1","slug":"evolving-reinforcement-learning-algorithms-1","title":"Evolving Reinforcement Learning Algorithms","date":"2021-01-08","arxiv_id":"2101.03958","repositories_listed":5,"syntology":null},{"url":"/paper/acme-a-research-framework-for-distributed","slug":"acme-a-research-framework-for-distributed","title":"Acme: A Research Framework for Distributed Reinforcement Learning","date":"2020-06-01","arxiv_id":"2006.00979","repositories_listed":5,"syntology":null},{"url":"/paper/agent57-outperforming-the-atari-human","slug":"agent57-outperforming-the-atari-human","title":"Agent57: Outperforming the Atari Human Benchmark","date":"2020-03-30","arxiv_id":"2003.13350","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agent57-outperforming-the-atari-human#ran","syntology_url":"https://syntology.ai/paper/2003.13350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13350"}},"official":null}},{"url":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformers-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1910.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06764"}},"official":null}},{"url":"/paper/a-simple-randomization-technique-for-1","slug":"a-simple-randomization-technique-for-1","title":"Network Randomization: A Simple Technique for Generalization in Deep Reinforcement Learning","date":"2019-10-11","arxiv_id":"1910.05396","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-randomization-technique-for-1#ran","syntology_url":"https://syntology.ai/paper/1910.05396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05396"}},"official":{"repos":["pokaxpoka/netrand"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-batch-deep-reinforcement","slug":"benchmarking-batch-deep-reinforcement","title":"Benchmarking Batch Deep Reinforcement Learning Algorithms","date":"2019-10-03","arxiv_id":"1910.01708","repositories_listed":5,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-3","slug":"multi-agent-deep-reinforcement-learning-for-3","title":"Multi-Agent Deep Reinforcement Learning for Liquidation Strategy Analysis","date":"2019-06-24","arxiv_id":"1906.11046","repositories_listed":5,"syntology":null},{"url":"/paper/sqil-imitation-learning-via-regularized","slug":"sqil-imitation-learning-via-regularized","title":"SQIL: Imitation Learning via Reinforcement Learning with Sparse Rewards","date":"2019-05-27","arxiv_id":"1905.11108","repositories_listed":5,"syntology":null},{"url":"/paper/crossnorm-normalization-for-off-policy-td","slug":"crossnorm-normalization-for-off-policy-td","title":"CrossQ: Batch Normalization in Deep Reinforcement Learning for Greater Sample Efficiency and Simplicity","date":"2019-02-14","arxiv_id":"1902.05605","repositories_listed":5,"syntology":null},{"url":"/paper/decoupling-feature-extraction-from-policy","slug":"decoupling-feature-extraction-from-policy","title":"Decoupling feature extraction from policy learning: assessing benefits of state representation learning in goal based robotics","date":"2019-01-24","arxiv_id":"1901.08651","repositories_listed":5,"syntology":null},{"url":"/paper/learning-to-design-rna","slug":"learning-to-design-rna","title":"Learning to Design RNA","date":"2018-12-31","arxiv_id":"1812.11951","repositories_listed":5,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-design-rna#ran","syntology_url":"https://syntology.ai/paper/1812.11951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.11951"}},"official":{"repos":["automl/learna"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-introduction-to-deep-reinforcement","slug":"an-introduction-to-deep-reinforcement","title":"An Introduction to Deep Reinforcement Learning","date":"2018-11-30","arxiv_id":"1811.12560","repositories_listed":5,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based","slug":"deep-reinforcement-learning-based","title":"Deep Reinforcement Learning based Recommendation with Explicit User-Item Interactions Modeling","date":"2018-10-29","arxiv_id":"1810.12027","repositories_listed":5,"syntology":null},{"url":"/paper/deep-reinforcement-learning","slug":"deep-reinforcement-learning","title":"Deep Reinforcement Learning","date":"2018-10-15","arxiv_id":"1810.06339","repositories_listed":5,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06339","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06339"}},"official":{"repos":["deepmind/lab","google/dopamine","mrkulk/hierarchical-deep-RL","rllab/rllab"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/parametrized-deep-q-networks-learning","slug":"parametrized-deep-q-networks-learning","title":"Parametrized Deep Q-Networks Learning: Reinforcement Learning with Discrete-Continuous Hybrid Action Space","date":"2018-10-10","arxiv_id":"1810.06394","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parametrized-deep-q-networks-learning#ran","syntology_url":"https://syntology.ai/paper/1810.06394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06394"}},"official":null}},{"url":"/paper/variational-discriminator-bottleneck","slug":"variational-discriminator-bottleneck","title":"Variational Discriminator Bottleneck: Improving Imitation Learning, Inverse RL, and GANs by Constraining Information Flow","date":"2018-10-01","arxiv_id":"1810.00821","repositories_listed":5,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/variational-discriminator-bottleneck#ran","syntology_url":"https://syntology.ai/paper/1810.00821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.00821"}},"official":null}},{"url":"/paper/s-rl-toolbox-environments-datasets-and","slug":"s-rl-toolbox-environments-datasets-and","title":"S-RL Toolbox: Environments, Datasets and Evaluation Metrics for State Representation Learning","date":"2018-09-25","arxiv_id":"1809.09369","repositories_listed":5,"syntology":null},{"url":"/paper/adversarial-deep-reinforcement-learning-in","slug":"adversarial-deep-reinforcement-learning-in","title":"Adversarial Deep Reinforcement Learning in Portfolio Management","date":"2018-08-29","arxiv_id":"1808.09940","repositories_listed":5,"syntology":null},{"url":"/paper/neural-architecture-optimization","slug":"neural-architecture-optimization","title":"Neural Architecture Optimization","date":"2018-08-22","arxiv_id":"1808.07233","repositories_listed":5,"syntology":null},{"url":"/paper/large-scale-study-of-curiosity-driven","slug":"large-scale-study-of-curiosity-driven","title":"Large-Scale Study of Curiosity-Driven Learning","date":"2018-08-13","arxiv_id":"1808.04355","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-scale-study-of-curiosity-driven#ran","syntology_url":"https://syntology.ai/paper/1808.04355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.04355"}},"official":{"repos":["openai/large-scale-curiosity"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/distributed-distributional-deterministic","slug":"distributed-distributional-deterministic","title":"Distributed Distributional Deterministic Policy Gradients","date":"2018-04-23","arxiv_id":"1804.08617","repositories_listed":5,"syntology":null},{"url":"/paper/differentiable-plasticity-training-plastic","slug":"differentiable-plasticity-training-plastic","title":"Differentiable plasticity: training plastic neural networks with backpropagation","date":"2018-04-06","arxiv_id":"1804.02464","repositories_listed":5,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/differentiable-plasticity-training-plastic#ran","syntology_url":"https://syntology.ai/paper/1804.02464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.02464"}},"official":{"repos":["uber-common/differentiable-plasticity"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/reinforcement-learning-on-web-interfaces","slug":"reinforcement-learning-on-web-interfaces","title":"Reinforcement Learning on Web Interfaces Using Workflow-Guided Exploration","date":"2018-02-24","arxiv_id":"1802.08802","repositories_listed":5,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-on-web-interfaces#ran","syntology_url":"https://syntology.ai/paper/1802.08802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08802"}},"official":{"repos":["stanfordnlp/wge","worksheets.codalab.org/worksheets/0x0f25031bd42f4aabbc17625fe1484066"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fully-decentralized-multi-agent-reinforcement","slug":"fully-decentralized-multi-agent-reinforcement","title":"Fully Decentralized Multi-Agent Reinforcement Learning with Networked Agents","date":"2018-02-23","arxiv_id":"1802.08757","repositories_listed":5,"syntology":null},{"url":"/paper/model-compression-via-distillation-and","slug":"model-compression-via-distillation-and","title":"Model compression via distillation and quantization","date":"2018-02-15","arxiv_id":"1802.05668","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-compression-via-distillation-and#ran","syntology_url":"https://syntology.ai/paper/1802.05668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.05668"}},"official":{"repos":["antspy/quantized_distillation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dice-the-infinitely-differentiable-monte","slug":"dice-the-infinitely-differentiable-monte","title":"DiCE: The Infinitely Differentiable Monte-Carlo Estimator","date":"2018-02-14","arxiv_id":"1802.05098","repositories_listed":5,"syntology":null}],"record_sha256":"37768bda9b9e9e88ccea1771bbc7b3d62c79946ca34a04e503245bf9bde0bcf3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}