{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/model-based-reinforcement-learning/papers/ran/1","list_of":"/task/model-based-reinforcement-learning","task":"Model-based Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,72],"of":72,"counts":{"archive_papers_tagged":708,"with_a_code_link":234,"where_syntology_ran_a_sample":72,"not_listed_spam_title":0,"listed":708,"listed_where_code_ran":72,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":66,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":66,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/model-based-reinforcement-learning/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/on-rollouts-in-model-based-reinforcement","slug":"on-rollouts-in-model-based-reinforcement","title":"On Rollouts in Model-Based Reinforcement Learning","date":"2025-01-28","arxiv_id":"2501.16918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-rollouts-in-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2501.16918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16918"}},"official":{"repos":["data-science-in-mechanical-engineering/infoprop"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stealing-that-free-lunch-exposing-the-limits","slug":"stealing-that-free-lunch-exposing-the-limits","title":"Stealing That Free Lunch: Exposing the Limits of Dyna-Style Reinforcement Learning","date":"2024-12-18","arxiv_id":"2412.14312","repositories_listed":0,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stealing-that-free-lunch-exposing-the-limits#ran","syntology_url":"https://syntology.ai/paper/2412.14312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14312"}},"official":null}},{"url":"/paper/bayes-adaptive-monte-carlo-tree-search-for","slug":"bayes-adaptive-monte-carlo-tree-search-for","title":"Bayes Adaptive Monte Carlo Tree Search for Offline Model-based Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11234","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayes-adaptive-monte-carlo-tree-search-for#ran","syntology_url":"https://syntology.ai/paper/2410.11234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11234"}},"official":{"repos":["lucascjysdl/offline-rl-kit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/drama-mamba-enabled-model-based-reinforcement","slug":"drama-mamba-enabled-model-based-reinforcement","title":"Drama: Mamba-Enabled Model-Based Reinforcement Learning Is Sample and Parameter Efficient","date":"2024-10-11","arxiv_id":"2410.08893","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/drama-mamba-enabled-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2410.08893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08893"}},"official":{"repos":["realwenlongwang/drama"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-model-based-reinforcement-learning-1","slug":"robust-model-based-reinforcement-learning-1","title":"Robust Model-Based Reinforcement Learning with an Adversarial Auxiliary Model","date":"2024-06-14","arxiv_id":"2406.09976","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-model-based-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2406.09976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09976"}},"official":{"repos":["rmbpo-eval/rmbpo-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coprocessor-actor-critic-a-model-based","slug":"coprocessor-actor-critic-a-model-based","title":"Coprocessor Actor Critic: A Model-Based Reinforcement Learning Approach For Adaptive Brain Stimulation","date":"2024-06-10","arxiv_id":"2406.06714","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coprocessor-actor-critic-a-model-based#ran","syntology_url":"https://syntology.ai/paper/2406.06714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06714"}},"official":{"repos":["michelllepan/neural-coprocessors"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trust-the-model-where-it-trusts-itself-model","slug":"trust-the-model-where-it-trusts-itself-model","title":"Trust the Model Where It Trusts Itself -- Model-Based Actor-Critic with Uncertainty-Aware Rollout Adaption","date":"2024-05-29","arxiv_id":"2405.19014","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trust-the-model-where-it-trusts-itself-model#ran","syntology_url":"https://syntology.ai/paper/2405.19014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19014"}},"official":{"repos":["Data-Science-in-Mechanical-Engineering/macura"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ivideogpt-interactive-videogpts-are-scalable","slug":"ivideogpt-interactive-videogpts-are-scalable","title":"iVideoGPT: Interactive VideoGPTs are Scalable World Models","date":"2024-05-24","arxiv_id":"2405.15223","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ivideogpt-interactive-videogpts-are-scalable#ran","syntology_url":"https://syntology.ai/paper/2405.15223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15223"}},"official":{"repos":["thuml/iVideoGPT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-code-world-models-with-large","slug":"generating-code-world-models-with-large","title":"Generating Code World Models with Large Language Models Guided by Monte Carlo Tree Search","date":"2024-05-24","arxiv_id":"2405.15383","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-code-world-models-with-large#ran","syntology_url":"https://syntology.ai/paper/2405.15383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15383"}},"official":{"repos":["nicoladainese96/code-world-models"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-reinforcement-learning-for-7","slug":"model-based-reinforcement-learning-for-7","title":"Model-based Reinforcement Learning for Parameterized Action Spaces","date":"2024-04-03","arxiv_id":"2404.03037","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-7#ran","syntology_url":"https://syntology.ai/paper/2404.03037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03037"}},"official":{"repos":["valarzz/dlpa","valarzz/model-based-reinforcement-learning-for-parameterized-action-spaces"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sindy-rl-interpretable-and-efficient-model","slug":"sindy-rl-interpretable-and-efficient-model","title":"SINDy-RL: Interpretable and Efficient Model-Based Reinforcement Learning","date":"2024-03-14","arxiv_id":"2403.09110","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sindy-rl-interpretable-and-efficient-model#ran","syntology_url":"https://syntology.ai/paper/2403.09110","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09110"}},"official":{"repos":["nzolman/sindy-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mastering-memory-tasks-with-world-models","slug":"mastering-memory-tasks-with-world-models","title":"Mastering Memory Tasks with World Models","date":"2024-03-07","arxiv_id":"2403.04253","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/mastering-memory-tasks-with-world-models#ran","syntology_url":"https://syntology.ai/paper/2403.04253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04253"}},"official":{"repos":["chandar-lab/Recall2Imagine"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/a-distributional-analogue-to-the-successor","slug":"a-distributional-analogue-to-the-successor","title":"A Distributional Analogue to the Successor Representation","date":"2024-02-13","arxiv_id":"2402.08530","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-distributional-analogue-to-the-successor#ran","syntology_url":"https://syntology.ai/paper/2402.08530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08530"}},"official":{"repos":["jessefarebro/distributional-sr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/smx-sequential-monte-carlo-planning-for","slug":"smx-sequential-monte-carlo-planning-for","title":"SPO: Sequential Monte Carlo Policy Optimisation","date":"2024-02-12","arxiv_id":"2402.07963","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/smx-sequential-monte-carlo-planning-for#ran","syntology_url":"https://syntology.ai/paper/2402.07963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07963"}},"official":null}},{"url":"/paper/covo-mpc-theoretical-analysis-of-sampling","slug":"covo-mpc-theoretical-analysis-of-sampling","title":"CoVO-MPC: Theoretical Analysis of Sampling-based MPC and Optimal Covariance Design","date":"2024-01-14","arxiv_id":"2401.07369","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covo-mpc-theoretical-analysis-of-sampling#ran","syntology_url":"https://syntology.ai/paper/2401.07369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07369"}},"official":{"repos":["LeCAR-Lab/CoVO-MPC"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regularity-as-intrinsic-reward-for-free-play-1","slug":"regularity-as-intrinsic-reward-for-free-play-1","title":"Regularity as Intrinsic Reward for Free Play","date":"2023-12-03","arxiv_id":"2312.01473","repositories_listed":0,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":16,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/regularity-as-intrinsic-reward-for-free-play-1#ran","syntology_url":"https://syntology.ai/paper/2312.01473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01473"}},"official":null}},{"url":"/paper/td-mpc2-scalable-robust-world-models-for","slug":"td-mpc2-scalable-robust-world-models-for","title":"TD-MPC2: Scalable, Robust World Models for Continuous Control","date":"2023-10-25","arxiv_id":"2310.16828","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":4,"n_ran_checked":6,"n_instrument":6,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"12 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/td-mpc2-scalable-robust-world-models-for#ran","syntology_url":"https://syntology.ai/paper/2310.16828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16828"}},"official":null}},{"url":"/paper/storm-efficient-stochastic-transformer-based-1","slug":"storm-efficient-stochastic-transformer-based-1","title":"STORM: Efficient Stochastic Transformer based World Models for Reinforcement Learning","date":"2023-10-14","arxiv_id":"2310.09615","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/storm-efficient-stochastic-transformer-based-1#ran","syntology_url":"https://syntology.ai/paper/2310.09615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09615"}},"official":{"repos":["weipu-zhang/storm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-agent-tree-search-unifies-reasoning","slug":"language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","arxiv_id":"2310.04406","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-agent-tree-search-unifies-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.04406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04406"}},"official":{"repos":["lapisrocks/languageagenttreesearch","andyz245/LanguageAgentTreeSearch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/combining-spatial-and-temporal-abstraction-in","slug":"combining-spatial-and-temporal-abstraction-in","title":"Consciousness-Inspired Spatio-Temporal Abstractions for Better Generalization in Reinforcement Learning","date":"2023-09-30","arxiv_id":"2310.00229","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/combining-spatial-and-temporal-abstraction-in#ran","syntology_url":"https://syntology.ai/paper/2310.00229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00229"}},"official":{"repos":["mila-iqia/skipper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harmony-world-models-boosting-sample","slug":"harmony-world-models-boosting-sample","title":"HarmonyDream: Task Harmonization Inside World Models","date":"2023-09-30","arxiv_id":"2310.00344","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/harmony-world-models-boosting-sample#ran","syntology_url":"https://syntology.ai/paper/2310.00344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00344"}},"official":{"repos":["thuml/harmonydream"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-contextualized-world-models-with-1","slug":"pre-training-contextualized-world-models-with-1","title":"Pre-training Contextualized World Models with In-the-wild Videos for Reinforcement Learning","date":"2023-05-29","arxiv_id":"2305.18499","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pre-training-contextualized-world-models-with-1#ran","syntology_url":"https://syntology.ai/paper/2305.18499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18499"}},"official":{"repos":["thuml/ContextWM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/flex-an-adaptive-exploration-algorithm-for","slug":"flex-an-adaptive-exploration-algorithm-for","title":"FLEX: an Adaptive Exploration Algorithm for Nonlinear Systems","date":"2023-04-26","arxiv_id":"2304.13426","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flex-an-adaptive-exploration-algorithm-for#ran","syntology_url":"https://syntology.ai/paper/2304.13426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13426"}},"official":{"repos":["mb-29/exploration"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-update-to-data-ratio-minimizing-world","slug":"dynamic-update-to-data-ratio-minimizing-world","title":"Dynamic Update-to-Data Ratio: Minimizing World Model Overfitting","date":"2023-03-17","arxiv_id":"2303.10144","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-update-to-data-ratio-minimizing-world#ran","syntology_url":"https://syntology.ai/paper/2303.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10144"}},"official":{"repos":["nicolinho/dutd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-world-models-are-happy-with","slug":"transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","repositories_listed":1,"syntology":{"n":25,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformer-based-world-models-are-happy-with#ran","syntology_url":"https://syntology.ai/paper/2303.07109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07109"}},"official":{"repos":["jrobine/twm"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/predictive-experience-replay-for-continual","slug":"predictive-experience-replay-for-continual","title":"Predictive Experience Replay for Continual Visual Control and Forecasting","date":"2023-03-12","arxiv_id":"2303.06572","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/predictive-experience-replay-for-continual#ran","syntology_url":"https://syntology.ai/paper/2303.06572","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06572"}},"official":{"repos":["wendongzh/continual_visual_control","jc043/cpl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-decentralized-cross-entropy-method","slug":"a-simple-decentralized-cross-entropy-method","title":"A Simple Decentralized Cross-Entropy Method","date":"2022-12-16","arxiv_id":"2212.08235","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-simple-decentralized-cross-entropy-method#ran","syntology_url":"https://syntology.ai/paper/2212.08235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08235"}},"official":{"repos":["vincentzhang/decentcem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-benefits-of-model-based-generalization-in","slug":"the-benefits-of-model-based-generalization-in","title":"The Benefits of Model-Based Generalization in Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02222","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-benefits-of-model-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2211.02222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02222"}},"official":{"repos":["kenjyoung/model_generalization_code_supplement"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-feasibility-of-cross-task-transfer","slug":"on-the-feasibility-of-cross-task-transfer","title":"On the Feasibility of Cross-Task Transfer with Model-Based Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10763","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/on-the-feasibility-of-cross-task-transfer#ran","syntology_url":"https://syntology.ai/paper/2210.10763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10763"}},"official":{"repos":["mlpc-ucsd/xtra"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-model-based-reinforcement-learning-with-2","slug":"safe-model-based-reinforcement-learning-with-2","title":"Safe Model-Based Reinforcement Learning with an Uncertainty-Aware Reachability Certificate","date":"2022-10-14","arxiv_id":"2210.07553","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-model-based-reinforcement-learning-with-2#ran","syntology_url":"https://syntology.ai/paper/2210.07553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07553"}},"official":{"repos":["ManUtdMoon/Safe_MBRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-alternating-offline","slug":"a-unified-framework-for-alternating-offline","title":"A Unified Framework for Alternating Offline Model Training and Policy Learning","date":"2022-10-12","arxiv_id":"2210.05922","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-alternating-offline#ran","syntology_url":"https://syntology.ai/paper/2210.05922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05922"}},"official":{"repos":["shentao-yang/ampl_neurips2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/live-in-the-moment-learning-dynamics-model","slug":"live-in-the-moment-learning-dynamics-model","title":"Live in the Moment: Learning Dynamics Model Adapted to Evolving Policy","date":"2022-07-25","arxiv_id":"2207.12141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/live-in-the-moment-learning-dynamics-model#ran","syntology_url":"https://syntology.ai/paper/2207.12141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12141"}},"official":{"repos":["si0wang/pdml"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/causal-dynamics-learning-for-task-independent","slug":"causal-dynamics-learning-for-task-independent","title":"Causal Dynamics Learning for Task-Independent State Abstraction","date":"2022-06-27","arxiv_id":"2206.13452","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/causal-dynamics-learning-for-task-independent#ran","syntology_url":"https://syntology.ai/paper/2206.13452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13452"}},"official":{"repos":["wangzizhao/causaldynamicslearning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/transdreamer-reinforcement-learning-with-1","slug":"transdreamer-reinforcement-learning-with-1","title":"TransDreamer: Reinforcement Learning with Transformer World Models","date":"2022-02-19","arxiv_id":"2202.09481","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transdreamer-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2202.09481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.09481"}},"official":null}},{"url":"/paper/exponential-family-model-based-reinforcement","slug":"exponential-family-model-based-reinforcement","title":"Exponential Family Model-Based Reinforcement Learning via Score Matching","date":"2021-12-28","arxiv_id":"2112.14195","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exponential-family-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.14195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.14195"}},"official":{"repos":["anmolkabra/score-matching-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-experimental-design-perspective-on-model","slug":"an-experimental-design-perspective-on-model","title":"An Experimental Design Perspective on Model-Based Reinforcement Learning","date":"2021-12-09","arxiv_id":"2112.05244","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-experimental-design-perspective-on-model#ran","syntology_url":"https://syntology.ai/paper/2112.05244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05244"}},"official":null}},{"url":"/paper/sample-complexity-of-robust-reinforcement","slug":"sample-complexity-of-robust-reinforcement","title":"Sample Complexity of Robust Reinforcement Learning with a Generative Model","date":"2021-12-02","arxiv_id":"2112.01506","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-complexity-of-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.01506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.01506"}},"official":{"repos":["kishanpb/RobustRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dreamerpro-reconstruction-free-model-based-1","slug":"dreamerpro-reconstruction-free-model-based-1","title":"DreamerPro: Reconstruction-Free Model-Based Reinforcement Learning with Prototypical Representations","date":"2021-10-27","arxiv_id":"2110.14565","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dreamerpro-reconstruction-free-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2110.14565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14565"}},"official":{"repos":["fdeng18/dreamer-pro"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pc-mlp-model-based-reinforcement-learning","slug":"pc-mlp-model-based-reinforcement-learning","title":"PC-MLP: Model-based Reinforcement Learning with Policy Cover Guided Exploration","date":"2021-07-15","arxiv_id":"2107.07410","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pc-mlp-model-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2107.07410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07410"}},"official":{"repos":["yudasong/PCMLP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/systematic-evaluation-of-causal-discovery-in-1","slug":"systematic-evaluation-of-causal-discovery-in-1","title":"Systematic Evaluation of Causal Discovery in Visual Model Based Reinforcement Learning","date":"2021-07-02","arxiv_id":"2107.00848","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/systematic-evaluation-of-causal-discovery-in-1#ran","syntology_url":"https://syntology.ai/paper/2107.00848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00848"}},"official":{"repos":["dido1998/CausalMBRL"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-predictive-coding-for-model-based","slug":"temporal-predictive-coding-for-model-based","title":"Temporal Predictive Coding For Model-Based Planning In Latent Space","date":"2021-06-14","arxiv_id":"2106.07156","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/temporal-predictive-coding-for-model-based#ran","syntology_url":"https://syntology.ai/paper/2106.07156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.07156"}},"official":{"repos":["VinAIResearch/TPC-tensorflow","tung-nd/TPC-tensorflow"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/control-oriented-model-based-reinforcement","slug":"control-oriented-model-based-reinforcement","title":"Control-Oriented Model-Based Reinforcement Learning with Implicit Differentiation","date":"2021-06-06","arxiv_id":"2106.03273","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/control-oriented-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03273"}},"official":{"repos":["evgenii-nikishin/omd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-consciousness-inspired-planning-agent-for","slug":"a-consciousness-inspired-planning-agent-for","title":"A Consciousness-Inspired Planning Agent for Model-Based Reinforcement Learning","date":"2021-06-03","arxiv_id":"2106.02097","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-consciousness-inspired-planning-agent-for#ran","syntology_url":"https://syntology.ai/paper/2106.02097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02097"}},"official":{"repos":["mila-iqia/conscious-planning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/learning-modular-robot-control-policies","slug":"learning-modular-robot-control-policies","title":"Learning Modular Robot Control Policies","date":"2021-05-20","arxiv_id":"2105.10049","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-modular-robot-control-policies#ran","syntology_url":"https://syntology.ai/paper/2105.10049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10049"}},"official":{"repos":["biorobotics/learning_modular_policies"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-drive-from-a-world-on-rails","slug":"learning-to-drive-from-a-world-on-rails","title":"Learning to drive from a world on rails","date":"2021-05-03","arxiv_id":"2105.00636","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-drive-from-a-world-on-rails#ran","syntology_url":"https://syntology.ai/paper/2105.00636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.00636"}},"official":{"repos":["dotchen/WorldOnRails"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-predictive-actor-critic-accelerating","slug":"model-predictive-actor-critic-accelerating","title":"Model Predictive Actor-Critic: Accelerating Robot Skill Acquisition with Deep Reinforcement Learning","date":"2021-03-25","arxiv_id":"2103.13842","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-predictive-actor-critic-accelerating#ran","syntology_url":"https://syntology.ai/paper/2103.13842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13842"}},"official":{"repos":["dnandha/mopac"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-wise-multiple-choice-learning-for","slug":"trajectory-wise-multiple-choice-learning-for","title":"Trajectory-wise Multiple Choice Learning for Dynamics Generalization in Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13303","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-wise-multiple-choice-learning-for#ran","syntology_url":"https://syntology.ai/paper/2010.13303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13303"}},"official":{"repos":["younggyoseo/trajectory_mcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-imagination-and-reality-for-model","slug":"bridging-imagination-and-reality-for-model","title":"Bridging Imagination and Reality for Model-Based Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12142","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-imagination-and-reality-for-model#ran","syntology_url":"https://syntology.ai/paper/2010.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12142"}},"official":{"repos":["Mehooz/BIRD_code"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/model-based-policy-optimization-with","slug":"model-based-policy-optimization-with","title":"Model-based Policy Optimization with Unsupervised Model Adaptation","date":"2020-10-19","arxiv_id":"2010.09546","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/model-based-policy-optimization-with#ran","syntology_url":"https://syntology.ai/paper/2010.09546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09546"}},"official":{"repos":["RockySJ/ampo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dynode-neural-ordinary-differential-equations","slug":"dynode-neural-ordinary-differential-equations","title":"DyNODE: Neural Ordinary Differential Equations for Dynamics Modeling in Continuous Control","date":"2020-09-09","arxiv_id":"2009.04278","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynode-neural-ordinary-differential-equations#ran","syntology_url":"https://syntology.ai/paper/2009.04278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.04278"}},"official":{"repos":["vmartinezalvarez/DyNODE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-cross-entropy-method-for","slug":"sample-efficient-cross-entropy-method-for","title":"Sample-efficient Cross-Entropy Method for Real-time Planning","date":"2020-08-14","arxiv_id":"2008.06389","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sample-efficient-cross-entropy-method-for#ran","syntology_url":"https://syntology.ai/paper/2008.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06389"}},"official":{"repos":["martius-lab/iCEM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-loca-regret-a-consistent-metric-to","slug":"the-loca-regret-a-consistent-metric-to","title":"The LoCA Regret: A Consistent Metric to Evaluate Model-Based Behavior in Reinforcement Learning","date":"2020-07-07","arxiv_id":"2007.03158","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/the-loca-regret-a-consistent-metric-to#ran","syntology_url":"https://syntology.ai/paper/2007.03158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03158"}},"official":{"repos":["chandar-lab/LoCA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bidirectional-model-based-policy-optimization","slug":"bidirectional-model-based-policy-optimization","title":"Bidirectional Model-based Policy Optimization","date":"2020-07-04","arxiv_id":"2007.01995","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bidirectional-model-based-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2007.01995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01995"}},"official":{"repos":["hanglai/bmpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/context-aware-dynamics-model-for","slug":"context-aware-dynamics-model-for","title":"Context-aware Dynamics Model for Generalization in Model-Based Reinforcement Learning","date":"2020-05-14","arxiv_id":"2005.06800","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/context-aware-dynamics-model-for#ran","syntology_url":"https://syntology.ai/paper/2005.06800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.06800"}},"official":{"repos":["younggyoseo/CaDM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-game-theoretic-framework-for-model-based","slug":"a-game-theoretic-framework-for-model-based","title":"A Game Theoretic Framework for Model Based Reinforcement Learning","date":"2020-04-16","arxiv_id":"2004.07804","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-game-theoretic-framework-for-model-based#ran","syntology_url":"https://syntology.ai/paper/2004.07804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07804"}},"official":null}},{"url":"/paper/a-model-based-reinforcement-learning-with","slug":"a-model-based-reinforcement-learning-with","title":"Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-11-10","arxiv_id":"1911.03845","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-model-based-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/1911.03845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.03845"}},"official":{"repos":["JianGuanTHU/IRecGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-predict-without-looking-ahead","slug":"learning-to-predict-without-looking-ahead","title":"Learning to Predict Without Looking Ahead: World Models Without Forward Prediction","date":"2019-10-29","arxiv_id":"1910.13038","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-predict-without-looking-ahead#ran","syntology_url":"https://syntology.ai/paper/1910.13038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.13038"}},"official":null}},{"url":"/paper/entity-abstraction-in-visual-model-based","slug":"entity-abstraction-in-visual-model-based","title":"Entity Abstraction in Visual Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/entity-abstraction-in-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/1910.12827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12827"}},"official":{"repos":["jcoreyes/OP3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamics-aware-unsupervised-discovery-of","slug":"dynamics-aware-unsupervised-discovery-of","title":"Dynamics-Aware Unsupervised Discovery of Skills","date":"2019-07-02","arxiv_id":"1907.01657","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamics-aware-unsupervised-discovery-of#ran","syntology_url":"https://syntology.ai/paper/1907.01657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01657"}},"official":{"repos":["google-research/dads"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-to-trust-your-model-model-based-policy","slug":"when-to-trust-your-model-model-based-policy","title":"When to Trust Your Model: Model-Based Policy Optimization","date":"2019-06-19","arxiv_id":"1906.08253","repositories_listed":11,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-to-trust-your-model-model-based-policy#ran","syntology_url":"https://syntology.ai/paper/1906.08253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.08253"}},"official":{"repos":["JannerM/mbpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/extending-deep-model-predictive-control-with","slug":"extending-deep-model-predictive-control-with","title":"Safety Augmented Value Estimation from Demonstrations (SAVED): Safe Deep Model-Based RL for Sparse Cost Robotic Tasks","date":"2019-05-31","arxiv_id":"1905.13402","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extending-deep-model-predictive-control-with#ran","syntology_url":"https://syntology.ai/paper/1905.13402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13402"}},"official":null}},{"url":"/paper/tight-regret-bounds-for-model-based","slug":"tight-regret-bounds-for-model-based","title":"Tight Regret Bounds for Model-Based Reinforcement Learning with Greedy Policies","date":"2019-05-27","arxiv_id":"1905.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tight-regret-bounds-for-model-based#ran","syntology_url":"https://syntology.ai/paper/1905.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11527"}},"official":{"repos":["NMerlis/TabulaRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-reinforcement-learning-for-atari","slug":"model-based-reinforcement-learning-for-atari","title":"Model-Based Reinforcement Learning for Atari","date":"2019-03-01","arxiv_id":"1903.00374","repositories_listed":2,"syntology":{"n":21,"n_ran":16,"n_constructed":8,"n_ran_checked":10,"n_instrument":6,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"16 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-atari#ran","syntology_url":"https://syntology.ai/paper/1903.00374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.00374"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-general-framework-for-structured-learning","slug":"a-general-framework-for-structured-learning","title":"A General Framework for Structured Learning of Mechanical Systems","date":"2019-02-22","arxiv_id":"1902.08705","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-general-framework-for-structured-learning#ran","syntology_url":"https://syntology.ai/paper/1902.08705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.08705"}},"official":{"repos":["sisl/mechamodlearn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/algorithmic-framework-for-model-based-deep","slug":"algorithmic-framework-for-model-based-deep","title":"Algorithmic Framework for Model-based Deep Reinforcement Learning with Theoretical Guarantees","date":"2018-07-10","arxiv_id":"1807.03858","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/algorithmic-framework-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1807.03858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03858"}},"official":{"repos":["roosephu/slbo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-in-a-handful-of","slug":"deep-reinforcement-learning-in-a-handful-of","title":"Deep Reinforcement Learning in a Handful of Trials using Probabilistic Dynamics Models","date":"2018-05-30","arxiv_id":"1805.12114","repositories_listed":9,"syntology":{"n":19,"n_ran":12,"n_constructed":5,"n_ran_checked":7,"n_instrument":5,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":17,"phrase":"12 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/deep-reinforcement-learning-in-a-handful-of#ran","syntology_url":"https://syntology.ai/paper/1805.12114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.12114"}},"official":{"repos":["kchua/handful-of-trials"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/neural-network-dynamics-for-model-based-deep","slug":"neural-network-dynamics-for-model-based-deep","title":"Neural Network Dynamics for Model-Based Deep Reinforcement Learning with Model-Free Fine-Tuning","date":"2017-08-08","arxiv_id":"1708.02596","repositories_listed":9,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-network-dynamics-for-model-based-deep#ran","syntology_url":"https://syntology.ai/paper/1708.02596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1708.02596"}},"official":null}},{"url":"/paper/imagination-augmented-agents-for-deep","slug":"imagination-augmented-agents-for-deep","title":"Imagination-Augmented Agents for Deep Reinforcement Learning","date":"2017-07-19","arxiv_id":"1707.06203","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imagination-augmented-agents-for-deep#ran","syntology_url":"https://syntology.ai/paper/1707.06203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1707.06203"}},"official":null}}],"record_sha256":"40261597ac2b211f9d7d6ce27332dcd4bde93e423746ec972af372e87f3f3aa7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}