{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/model-based-reinforcement-learning/papers/2","list_of":"/task/model-based-reinforcement-learning","task":"Model-based Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":8,"rows_per_page":100,"rows":[101,200],"of":708,"counts":{"archive_papers_tagged":708,"with_a_code_link":234,"where_syntology_ran_a_sample":72,"not_listed_spam_title":0,"listed":708,"listed_where_code_ran":72,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":66,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":66,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/model-based-reinforcement-learning","prev":"/task/model-based-reinforcement-learning","next":"/task/model-based-reinforcement-learning/papers/3","papers":[{"url":"/paper/semi-infinitely-constrained-markov-decision","slug":"semi-infinitely-constrained-markov-decision","title":"Semi-Infinitely Constrained Markov Decision Processes and Efficient Reinforcement Learning","date":"2023-04-29","arxiv_id":"2305.00254","repositories_listed":1,"syntology":null},{"url":"/paper/flex-an-adaptive-exploration-algorithm-for","slug":"flex-an-adaptive-exploration-algorithm-for","title":"FLEX: an Adaptive Exploration Algorithm for Nonlinear Systems","date":"2023-04-26","arxiv_id":"2304.13426","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flex-an-adaptive-exploration-algorithm-for#ran","syntology_url":"https://syntology.ai/paper/2304.13426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13426"}},"official":{"repos":["mb-29/exploration"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-model-based-reinforcement","slug":"sample-efficient-model-based-reinforcement","title":"Sample-efficient Model-based Reinforcement Learning for Quantum Control","date":"2023-04-19","arxiv_id":"2304.09718","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-with-3","slug":"model-based-reinforcement-learning-with-3","title":"Model-Based Reinforcement Learning with Isolated Imaginations","date":"2023-03-27","arxiv_id":"2303.14889","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-update-to-data-ratio-minimizing-world","slug":"dynamic-update-to-data-ratio-minimizing-world","title":"Dynamic Update-to-Data Ratio: Minimizing World Model Overfitting","date":"2023-03-17","arxiv_id":"2303.10144","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-update-to-data-ratio-minimizing-world#ran","syntology_url":"https://syntology.ai/paper/2303.10144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.10144"}},"official":{"repos":["nicolinho/dutd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-based-world-models-are-happy-with","slug":"transformer-based-world-models-are-happy-with","title":"Transformer-based World Models Are Happy With 100k Interactions","date":"2023-03-13","arxiv_id":"2303.07109","repositories_listed":1,"syntology":{"n":25,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":9,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 8 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformer-based-world-models-are-happy-with#ran","syntology_url":"https://syntology.ai/paper/2303.07109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07109"}},"official":{"repos":["jrobine/twm"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/diminishing-return-of-value-expansion-methods","slug":"diminishing-return-of-value-expansion-methods","title":"Diminishing Return of Value Expansion Methods in Model-Based Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03955","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diminishing-return-of-value-expansion-methods#ran","syntology_url":"https://syntology.ai/paper/2303.03955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03955"}},"official":{"repos":["danielpalen/value_expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-real-time-planning-with","slug":"sample-efficient-real-time-planning-with","title":"Sample-efficient Real-time Planning with Curiosity Cross-Entropy Method and Contrastive Learning","date":"2023-03-07","arxiv_id":"2303.03787","repositories_listed":1,"syntology":null},{"url":"/paper/the-virtues-of-laziness-in-model-based-rl-a","slug":"the-virtues-of-laziness-in-model-based-rl-a","title":"The Virtues of Laziness in Model-based RL: A Unified Objective and Algorithms","date":"2023-03-01","arxiv_id":"2303.00694","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/non-zero-sum-game-control-for-multi-vehicle","slug":"non-zero-sum-game-control-for-multi-vehicle","title":"Non-zero-sum Game Control for Multi-vehicle Driving via Reinforcement Learning","date":"2023-02-08","arxiv_id":"2302.03958","repositories_listed":1,"syntology":null},{"url":"/paper/plan-to-predict-learning-an-uncertainty","slug":"plan-to-predict-learning-an-uncertainty","title":"Plan To Predict: Learning an Uncertainty-Foreseeing Model for Model-Based Reinforcement Learning","date":"2023-01-20","arxiv_id":"2301.08502","repositories_listed":1,"syntology":null},{"url":"/paper/hint-assisted-reinforcement-learning-an","slug":"hint-assisted-reinforcement-learning-an","title":"Hint assisted reinforcement learning: an application in radio astronomy","date":"2023-01-10","arxiv_id":"2301.03933","repositories_listed":1,"syntology":null},{"url":"/paper/pontryagin-optimal-controller-via-neural","slug":"pontryagin-optimal-controller-via-neural","title":"Pontryagin Optimal Control via Neural Networks","date":"2022-12-30","arxiv_id":"2212.14566","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-decentralized-cross-entropy-method","slug":"a-simple-decentralized-cross-entropy-method","title":"A Simple Decentralized Cross-Entropy Method","date":"2022-12-16","arxiv_id":"2212.08235","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-simple-decentralized-cross-entropy-method#ran","syntology_url":"https://syntology.ai/paper/2212.08235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08235"}},"official":{"repos":["vincentzhang/decentcem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/modem-accelerating-visual-model-based","slug":"modem-accelerating-visual-model-based","title":"MoDem: Accelerating Visual Model-Based Reinforcement Learning with Demonstrations","date":"2022-12-12","arxiv_id":"2212.05698","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/modem-accelerating-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/2212.05698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05698"}},"official":{"repos":["facebookresearch/modem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/physics-informed-model-based-reinforcement","slug":"physics-informed-model-based-reinforcement","title":"Physics-Informed Model-Based Reinforcement Learning","date":"2022-12-05","arxiv_id":"2212.02179","repositories_listed":1,"syntology":null},{"url":"/paper/the-benefits-of-model-based-generalization-in","slug":"the-benefits-of-model-based-generalization-in","title":"The Benefits of Model-Based Generalization in Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02222","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-benefits-of-model-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2211.02222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02222"}},"official":{"repos":["kenjyoung/model_generalization_code_supplement"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-safety-in-model-based-reinforcement","slug":"learning-safety-in-model-based-reinforcement","title":"Learning safety in model-based Reinforcement Learning using MPC and Gaussian Processes","date":"2022-11-03","arxiv_id":"2211.01860","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-feasibility-of-cross-task-transfer","slug":"on-the-feasibility-of-cross-task-transfer","title":"On the Feasibility of Cross-Task Transfer with Model-Based Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.10763","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/on-the-feasibility-of-cross-task-transfer#ran","syntology_url":"https://syntology.ai/paper/2210.10763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10763"}},"official":{"repos":["mlpc-ucsd/xtra"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/on-uncertainty-in-deep-state-space-models-for","slug":"on-uncertainty-in-deep-state-space-models-for","title":"On Uncertainty in Deep State Space Models for Model-Based Reinforcement Learning","date":"2022-10-17","arxiv_id":"2210.09256","repositories_listed":1,"syntology":null},{"url":"/paper/when-to-update-your-model-constrained-model","slug":"when-to-update-your-model-constrained-model","title":"When to Update Your Model: Constrained Model-based Reinforcement Learning","date":"2022-10-15","arxiv_id":"2210.08349","repositories_listed":1,"syntology":null},{"url":"/paper/safe-model-based-reinforcement-learning-with-2","slug":"safe-model-based-reinforcement-learning-with-2","title":"Safe Model-Based Reinforcement Learning with an Uncertainty-Aware Reachability Certificate","date":"2022-10-14","arxiv_id":"2210.07553","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-model-based-reinforcement-learning-with-2#ran","syntology_url":"https://syntology.ai/paper/2210.07553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07553"}},"official":{"repos":["ManUtdMoon/Safe_MBRL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-unified-framework-for-alternating-offline","slug":"a-unified-framework-for-alternating-offline","title":"A Unified Framework for Alternating Offline Model Training and Policy Learning","date":"2022-10-12","arxiv_id":"2210.05922","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-alternating-offline#ran","syntology_url":"https://syntology.ai/paper/2210.05922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05922"}},"official":{"repos":["shentao-yang/ampl_neurips2022"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discover-deep-identification-of-symbolic-open","slug":"discover-deep-identification-of-symbolic-open","title":"DISCOVER: Deep identification of symbolically concise open-form PDEs via enhanced reinforcement-learning","date":"2022-10-04","arxiv_id":"2210.02181","repositories_listed":1,"syntology":null},{"url":"/paper/hyperbolic-vae-via-latent-gaussian-1","slug":"hyperbolic-vae-via-latent-gaussian-1","title":"Hyperbolic VAE via Latent Gaussian Distributions","date":"2022-09-30","arxiv_id":"2209.15217","repositories_listed":1,"syntology":null},{"url":"/paper/visuo-tactile-transformers-for-manipulation","slug":"visuo-tactile-transformers-for-manipulation","title":"Visuo-Tactile Transformers for Manipulation","date":"2022-09-30","arxiv_id":"2210.00121","repositories_listed":1,"syntology":null},{"url":"/paper/foresee-model-based-reinforcement-learning","slug":"foresee-model-based-reinforcement-learning","title":"FORESEE: Prediction with Expansion-Compression Unscented Transform for Online Policy Optimization","date":"2022-09-26","arxiv_id":"2209.12644","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-with-multi","slug":"model-based-reinforcement-learning-with-multi","title":"Model-based Reinforcement Learning with Multi-step Plan Value Estimation","date":"2022-09-12","arxiv_id":"2209.05530","repositories_listed":1,"syntology":null},{"url":"/paper/live-in-the-moment-learning-dynamics-model","slug":"live-in-the-moment-learning-dynamics-model","title":"Live in the Moment: Learning Dynamics Model Adapted to Evolving Policy","date":"2022-07-25","arxiv_id":"2207.12141","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/live-in-the-moment-learning-dynamics-model#ran","syntology_url":"https://syntology.ai/paper/2207.12141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12141"}},"official":{"repos":["si0wang/pdml"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/causal-dynamics-learning-for-task-independent","slug":"causal-dynamics-learning-for-task-independent","title":"Causal Dynamics Learning for Task-Independent State Abstraction","date":"2022-06-27","arxiv_id":"2206.13452","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/causal-dynamics-learning-for-task-independent#ran","syntology_url":"https://syntology.ai/paper/2206.13452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13452"}},"official":{"repos":["wangzizhao/causaldynamicslearning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-relational-intervention-approach-for-1","slug":"a-relational-intervention-approach-for-1","title":"A Relational Intervention Approach for Unsupervised Dynamics Generalization in Model-Based Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04551","repositories_listed":1,"syntology":null},{"url":"/paper/value-memory-graph-a-graph-structured-world","slug":"value-memory-graph-a-graph-structured-world","title":"Value Memory Graph: A Graph-Structured World Model for Offline Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04384","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-multi-agent-model-based","slug":"scalable-multi-agent-model-based","title":"Scalable Multi-Agent Model-Based Reinforcement Learning","date":"2022-05-25","arxiv_id":"2205.15023","repositories_listed":1,"syntology":null},{"url":"/paper/towards-biologically-plausible-dreaming-and","slug":"towards-biologically-plausible-dreaming-and","title":"Towards biologically plausible Dreaming and Planning in recurrent spiking networks","date":"2022-05-20","arxiv_id":"2205.10044","repositories_listed":1,"syntology":null},{"url":"/paper/rlflow-optimising-neural-network-subgraph","slug":"rlflow-optimising-neural-network-subgraph","title":"RLFlow: Optimising Neural Network Subgraph Transformation with World Models","date":"2022-05-03","arxiv_id":"2205.01435","repositories_listed":1,"syntology":null},{"url":"/paper/towards-evaluating-adaptivity-of-model-based","slug":"towards-evaluating-adaptivity-of-model-based","title":"Towards Evaluating Adaptivity of Model-Based Reinforcement Learning Methods","date":"2022-04-25","arxiv_id":"2204.11464","repositories_listed":1,"syntology":null},{"url":"/paper/learning-sequential-latent-variable-models","slug":"learning-sequential-latent-variable-models","title":"Learning Sequential Latent Variable Models from Multimodal Time Series Data","date":"2022-04-21","arxiv_id":"2204.10419","repositories_listed":1,"syntology":null},{"url":"/paper/value-gradient-weighted-model-based-1","slug":"value-gradient-weighted-model-based-1","title":"Value Gradient weighted Model-Based Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01464","repositories_listed":1,"syntology":null},{"url":"/paper/sage-generating-symbolic-goals-for-myopic","slug":"sage-generating-symbolic-goals-for-myopic","title":"SAGE: Generating Symbolic Goals for Myopic Models in Deep Reinforcement Learning","date":"2022-03-09","arxiv_id":"2203.05079","repositories_listed":1,"syntology":null},{"url":"/paper/learning-by-doing-controlling-a-dynamical","slug":"learning-by-doing-controlling-a-dynamical","title":"Learning by Doing: Controlling a Dynamical System using Causality, Control, and Reinforcement Learning","date":"2022-02-12","arxiv_id":"2202.06052","repositories_listed":1,"syntology":null},{"url":"/paper/bellman-meets-hawkes-model-based","slug":"bellman-meets-hawkes-model-based","title":"Bellman Meets Hawkes: Model-Based Reinforcement Learning via Temporal Point Processes","date":"2022-01-29","arxiv_id":"2201.12569","repositories_listed":1,"syntology":null},{"url":"/paper/exponential-family-model-based-reinforcement","slug":"exponential-family-model-based-reinforcement","title":"Exponential Family Model-Based Reinforcement Learning via Score Matching","date":"2021-12-28","arxiv_id":"2112.14195","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exponential-family-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.14195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.14195"}},"official":{"repos":["anmolkabra/score-matching-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cem-gd-cross-entropy-method-with-gradient","slug":"cem-gd-cross-entropy-method-with-gradient","title":"CEM-GD: Cross-Entropy Method with Gradient Descent Planner for Model-Based Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07746","repositories_listed":1,"syntology":null},{"url":"/paper/ed2-an-environment-dynamics-decomposition-1","slug":"ed2-an-environment-dynamics-decomposition-1","title":"ED2: Environment Dynamics Decomposition World Models for Continuous Control","date":"2021-12-06","arxiv_id":"2112.02817","repositories_listed":1,"syntology":null},{"url":"/paper/sample-complexity-of-robust-reinforcement","slug":"sample-complexity-of-robust-reinforcement","title":"Sample Complexity of Robust Reinforcement Learning with a Generative Model","date":"2021-12-02","arxiv_id":"2112.01506","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-complexity-of-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.01506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.01506"}},"official":{"repos":["kishanpb/RobustRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-state-representations-via-retracing-1","slug":"learning-state-representations-via-retracing-1","title":"Learning State Representations via Retracing in Reinforcement Learning","date":"2021-11-24","arxiv_id":"2111.12600","repositories_listed":1,"syntology":null},{"url":"/paper/on-effective-scheduling-of-model-based","slug":"on-effective-scheduling-of-model-based","title":"On Effective Scheduling of Model-based Reinforcement Learning","date":"2021-11-16","arxiv_id":"2111.08550","repositories_listed":1,"syntology":null},{"url":"/paper/dreamerpro-reconstruction-free-model-based-1","slug":"dreamerpro-reconstruction-free-model-based-1","title":"DreamerPro: Reconstruction-Free Model-Based Reinforcement Learning with Prototypical Representations","date":"2021-10-27","arxiv_id":"2110.14565","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dreamerpro-reconstruction-free-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2110.14565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14565"}},"official":{"repos":["fdeng18/dreamer-pro"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-model-based-reinforcement-learning-using","slug":"safe-model-based-reinforcement-learning-using","title":"Safe Reinforcement Learning Using Robust Control Barrier Functions","date":"2021-10-11","arxiv_id":"2110.05415","repositories_listed":1,"syntology":null},{"url":"/paper/mismatched-no-more-joint-model-policy","slug":"mismatched-no-more-joint-model-policy","title":"Mismatched No More: Joint Model-Policy Optimization for Model-Based RL","date":"2021-10-06","arxiv_id":"2110.02758","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-micro-data-reinforcement-learning-1","slug":"model-based-micro-data-reinforcement-learning-1","title":"Model-based micro-data reinforcement learning: what are the crucial model properties and which model to choose?","date":"2021-07-24","arxiv_id":"2107.11587","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/model-based-micro-data-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2107.11587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.11587"}},"official":{"repos":["ramp-kits/rl_simulator"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/know-thyself-transferable-visuomotor-control","slug":"know-thyself-transferable-visuomotor-control","title":"Know Thyself: Transferable Visual Control Policies Through Robot-Awareness","date":"2021-07-19","arxiv_id":"2107.09047","repositories_listed":1,"syntology":null},{"url":"/paper/pc-mlp-model-based-reinforcement-learning","slug":"pc-mlp-model-based-reinforcement-learning","title":"PC-MLP: Model-based Reinforcement Learning with Policy Cover Guided Exploration","date":"2021-07-15","arxiv_id":"2107.07410","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pc-mlp-model-based-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2107.07410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07410"}},"official":{"repos":["yudasong/PCMLP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/systematic-evaluation-of-causal-discovery-in-1","slug":"systematic-evaluation-of-causal-discovery-in-1","title":"Systematic Evaluation of Causal Discovery in Visual Model Based Reinforcement Learning","date":"2021-07-02","arxiv_id":"2107.00848","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/systematic-evaluation-of-causal-discovery-in-1#ran","syntology_url":"https://syntology.ai/paper/2107.00848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00848"}},"official":{"repos":["dido1998/CausalMBRL"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-task-informed-abstraction","slug":"learning-task-informed-abstraction","title":"Learning Task Informed Abstractions","date":"2021-06-29","arxiv_id":"2106.15612","repositories_listed":1,"syntology":null},{"url":"/paper/causal-reinforcement-learning-using","slug":"causal-reinforcement-learning-using","title":"Causal Reinforcement Learning using Observational and Interventional Data","date":"2021-06-28","arxiv_id":"2106.14421","repositories_listed":1,"syntology":null},{"url":"/paper/model-advantage-optimization-for-model-based","slug":"model-advantage-optimization-for-model-based","title":"Model-Advantage and Value-Aware Models for Model-Based Reinforcement Learning: Bridging the Gap in Theory and Practice","date":"2021-06-26","arxiv_id":"2106.14080","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-via-latent-1","slug":"model-based-reinforcement-learning-via-latent-1","title":"Model-Based Reinforcement Learning via Latent-Space Collocation","date":"2021-06-24","arxiv_id":"2106.13229","repositories_listed":1,"syntology":null},{"url":"/paper/proper-value-equivalence","slug":"proper-value-equivalence","title":"Proper Value Equivalence","date":"2021-06-18","arxiv_id":"2106.10316","repositories_listed":1,"syntology":null},{"url":"/paper/control-oriented-model-based-reinforcement","slug":"control-oriented-model-based-reinforcement","title":"Control-Oriented Model-Based Reinforcement Learning with Implicit Differentiation","date":"2021-06-06","arxiv_id":"2106.03273","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/control-oriented-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2106.03273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03273"}},"official":{"repos":["evgenii-nikishin/omd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-consciousness-inspired-planning-agent-for","slug":"a-consciousness-inspired-planning-agent-for","title":"A Consciousness-Inspired Planning Agent for Model-Based Reinforcement Learning","date":"2021-06-03","arxiv_id":"2106.02097","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-consciousness-inspired-planning-agent-for#ran","syntology_url":"https://syntology.ai/paper/2106.02097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02097"}},"official":{"repos":["mila-iqia/conscious-planning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/learning-neuro-symbolic-relational-transition","slug":"learning-neuro-symbolic-relational-transition","title":"Learning Neuro-Symbolic Relational Transition Models for Bilevel Planning","date":"2021-05-28","arxiv_id":"2105.14074","repositories_listed":1,"syntology":null},{"url":"/paper/learning-modular-robot-control-policies","slug":"learning-modular-robot-control-policies","title":"Learning Modular Robot Control Policies","date":"2021-05-20","arxiv_id":"2105.10049","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-modular-robot-control-policies#ran","syntology_url":"https://syntology.ai/paper/2105.10049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10049"}},"official":{"repos":["biorobotics/learning_modular_policies"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-optimization-in-bayesian-network","slug":"policy-optimization-in-bayesian-network","title":"Policy Optimization in Dynamic Bayesian Network Hybrid Models of Biomanufacturing Processes","date":"2021-05-13","arxiv_id":"2105.06543","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-drive-from-a-world-on-rails","slug":"learning-to-drive-from-a-world-on-rails","title":"Learning to drive from a world on rails","date":"2021-05-03","arxiv_id":"2105.00636","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-drive-from-a-world-on-rails#ran","syntology_url":"https://syntology.ai/paper/2105.00636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.00636"}},"official":{"repos":["dotchen/WorldOnRails"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-predictive-actor-critic-accelerating","slug":"model-predictive-actor-critic-accelerating","title":"Model Predictive Actor-Critic: Accelerating Robot Skill Acquisition with Deep Reinforcement Learning","date":"2021-03-25","arxiv_id":"2103.13842","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/model-predictive-actor-critic-accelerating#ran","syntology_url":"https://syntology.ai/paper/2103.13842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13842"}},"official":{"repos":["dnandha/mopac"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-importance-of-hyperparameter","slug":"on-the-importance-of-hyperparameter","title":"On the Importance of Hyperparameter Optimization for Model-based Reinforcement Learning","date":"2021-02-26","arxiv_id":"2102.13651","repositories_listed":1,"syntology":null},{"url":"/paper/visualizing-muzero-models","slug":"visualizing-muzero-models","title":"Visualizing MuZero Models","date":"2021-02-25","arxiv_id":"2102.12924","repositories_listed":1,"syntology":null},{"url":"/paper/towards-automatic-evaluation-of-dialog","slug":"towards-automatic-evaluation-of-dialog","title":"Towards Automatic Evaluation of Dialog Systems: A Model-Free Off-Policy Evaluation Approach","date":"2021-02-20","arxiv_id":"2102.10242","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-time-model-based-reinforcement","slug":"continuous-time-model-based-reinforcement","title":"Continuous-Time Model-Based Reinforcement Learning","date":"2021-02-09","arxiv_id":"2102.04764","repositories_listed":1,"syntology":null},{"url":"/paper/trust-but-verify-model-based-exploration-in","slug":"trust-but-verify-model-based-exploration-in","title":"Trust, but verify: model-based exploration in sparse reward environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-accurate-long-term-dynamics-for","slug":"learning-accurate-long-term-dynamics-for","title":"Learning Accurate Long-term Dynamics for Model-based Reinforcement Learning","date":"2020-12-16","arxiv_id":"2012.09156","repositories_listed":1,"syntology":null},{"url":"/paper/models-pixels-and-rewards-evaluating-design","slug":"models-pixels-and-rewards-evaluating-design","title":"Models, Pixels, and Rewards: Evaluating Design Trade-offs in Visual Model-Based Reinforcement Learning","date":"2020-12-08","arxiv_id":"2012.04603","repositories_listed":1,"syntology":null},{"url":"/paper/combining-semantic-guidance-and-deep","slug":"combining-semantic-guidance-and-deep","title":"Combining Semantic Guidance and Deep Reinforcement Learning For Generating Human Level Paintings","date":"2020-11-25","arxiv_id":"2011.12589","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayes-adaptive-deep-model-based-policy","slug":"bayes-adaptive-deep-model-based-policy","title":"Bayes-Adaptive Deep Model-Based Policy Optimisation","date":"2020-10-29","arxiv_id":"2010.15948","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-wise-multiple-choice-learning-for","slug":"trajectory-wise-multiple-choice-learning-for","title":"Trajectory-wise Multiple Choice Learning for Dynamics Generalization in Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13303","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-wise-multiple-choice-learning-for#ran","syntology_url":"https://syntology.ai/paper/2010.13303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13303"}},"official":{"repos":["younggyoseo/trajectory_mcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-imagination-and-reality-for-model","slug":"bridging-imagination-and-reality-for-model","title":"Bridging Imagination and Reality for Model-Based Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12142","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-imagination-and-reality-for-model#ran","syntology_url":"https://syntology.ai/paper/2010.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12142"}},"official":{"repos":["Mehooz/BIRD_code"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/dream-and-search-to-control-latent-space-1","slug":"dream-and-search-to-control-latent-space-1","title":"Dream and Search to Control: Latent Space Planning for Continuous Control","date":"2020-10-19","arxiv_id":"2010.09832","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-policy-optimization-with","slug":"model-based-policy-optimization-with","title":"Model-based Policy Optimization with Unsupervised Model Adaptation","date":"2020-10-19","arxiv_id":"2010.09546","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/model-based-policy-optimization-with#ran","syntology_url":"https://syntology.ai/paper/2010.09546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09546"}},"official":{"repos":["RockySJ/ampo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-model-based-reinforcement-learning-with-1","slug":"safe-model-based-reinforcement-learning-with-1","title":"Constrained Model-based Reinforcement Learning with Robust Cross-Entropy Method","date":"2020-10-15","arxiv_id":"2010.07968","repositories_listed":1,"syntology":null},{"url":"/paper/continual-model-based-reinforcement-learning","slug":"continual-model-based-reinforcement-learning","title":"Continual Model-Based Reinforcement Learning with Hypernetworks","date":"2020-09-25","arxiv_id":"2009.11997","repositories_listed":1,"syntology":null},{"url":"/paper/dynode-neural-ordinary-differential-equations","slug":"dynode-neural-ordinary-differential-equations","title":"DyNODE: Neural Ordinary Differential Equations for Dynamics Modeling in Continuous Control","date":"2020-09-09","arxiv_id":"2009.04278","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynode-neural-ordinary-differential-equations#ran","syntology_url":"https://syntology.ai/paper/2009.04278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.04278"}},"official":{"repos":["vmartinezalvarez/DyNODE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-model-based-stochastic-value-gradient","slug":"on-the-model-based-stochastic-value-gradient","title":"On the model-based stochastic value gradient for continuous reinforcement learning","date":"2020-08-28","arxiv_id":"2008.12775","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-cross-entropy-method-for","slug":"sample-efficient-cross-entropy-method-for","title":"Sample-efficient Cross-Entropy Method for Real-time Planning","date":"2020-08-14","arxiv_id":"2008.06389","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sample-efficient-cross-entropy-method-for#ran","syntology_url":"https://syntology.ai/paper/2008.06389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.06389"}},"official":{"repos":["martius-lab/iCEM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-abstract-models-for-strategic","slug":"learning-abstract-models-for-strategic","title":"Learning Abstract Models for Strategic Exploration and Fast Reward Transfer","date":"2020-07-12","arxiv_id":"2007.05896","repositories_listed":1,"syntology":null},{"url":"/paper/bidirectional-model-based-policy-optimization","slug":"bidirectional-model-based-policy-optimization","title":"Bidirectional Model-based Policy Optimization","date":"2020-07-04","arxiv_id":"2007.01995","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bidirectional-model-based-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2007.01995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.01995"}},"official":{"repos":["hanglai/bmpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-discretization-for-model-based","slug":"adaptive-discretization-for-model-based","title":"Adaptive Discretization for Model-Based Reinforcement Learning","date":"2020-07-01","arxiv_id":"2007.00717","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-reinforcement-learning-for-semi","slug":"model-based-reinforcement-learning-for-semi","title":"Model-based Reinforcement Learning for Semi-Markov Decision Processes with Neural ODEs","date":"2020-06-29","arxiv_id":"2006.16210","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/model-based-reinforcement-learning-for-semi#ran","syntology_url":"https://syntology.ai/paper/2006.16210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16210"}},"official":{"repos":["dtak/mbrl-smdp-ode"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/task-agnostic-online-reinforcement-learning","slug":"task-agnostic-online-reinforcement-learning","title":"Task-Agnostic Online Reinforcement Learning with an Infinite Mixture of Gaussian Processes","date":"2020-06-19","arxiv_id":"2006.11441","repositories_listed":1,"syntology":null},{"url":"/paper/delta-schema-network-in-model-based","slug":"delta-schema-network-in-model-based","title":"Delta Schema Network in Model-based Reinforcement Learning","date":"2020-06-17","arxiv_id":"2006.09950","repositories_listed":1,"syntology":null},{"url":"/paper/green-simulation-assisted-reinforcement","slug":"green-simulation-assisted-reinforcement","title":"Green Simulation Assisted Reinforcement Learning with Model Risk for Biomanufacturing Learning and Control","date":"2020-06-17","arxiv_id":"2006.09919","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-model-based-reinforcement-learning-1","slug":"efficient-model-based-reinforcement-learning-1","title":"Efficient Model-Based Reinforcement Learning through Optimistic Policy Search and Planning","date":"2020-06-15","arxiv_id":"2006.08684","repositories_listed":1,"syntology":null},{"url":"/paper/tools-for-data-driven-modeling-of-within-hand","slug":"tools-for-data-driven-modeling-of-within-hand","title":"Tools for Data-driven Modeling of Within-Hand Manipulation with Underactuated Adaptive Hands","date":"2020-06-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lifelong-control-of-off-grid-microgrid-with","slug":"lifelong-control-of-off-grid-microgrid-with","title":"Lifelong Control of Off-grid Microgrid with Model Based Reinforcement Learning","date":"2020-05-16","arxiv_id":"2005.08006","repositories_listed":1,"syntology":null},{"url":"/paper/delay-aware-model-based-reinforcement","slug":"delay-aware-model-based-reinforcement","title":"Delay-Aware Model-Based Reinforcement Learning for Continuous Control","date":"2020-05-11","arxiv_id":"2005.05440","repositories_listed":1,"syntology":null},{"url":"/paper/dynamics-aware-unsupervised-skill-discovery","slug":"dynamics-aware-unsupervised-skill-discovery","title":"Dynamics-Aware Unsupervised Skill Discovery","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/model-predictive-control-via-cross-entropy","slug":"model-predictive-control-via-cross-entropy","title":"Model-Predictive Control via Cross-Entropy and Gradient-Based Optimization","date":"2020-04-19","arxiv_id":"2004.08763","repositories_listed":1,"syntology":null},{"url":"/paper/neural-game-engine-accurate-learning","slug":"neural-game-engine-accurate-learning","title":"Neural Game Engine: Accurate learning of generalizable forward models from pixels","date":"2020-03-23","arxiv_id":"2003.10520","repositories_listed":1,"syntology":null}],"record_sha256":"b8ed61c171b4525cb942279c9fd6f6c7d1185a237b1fb81939b52e38bfd7978a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}