{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/model-based-reinforcement-learning/papers/3","list_of":"/task/model-based-reinforcement-learning","task":"Model-based Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":8,"rows_per_page":100,"rows":[201,300],"of":708,"counts":{"archive_papers_tagged":708,"with_a_code_link":234,"where_syntology_ran_a_sample":72,"not_listed_spam_title":0,"listed":708,"listed_where_code_ran":72,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":66,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":66,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/model-based-reinforcement-learning","prev":"/task/model-based-reinforcement-learning/papers/2","next":"/task/model-based-reinforcement-learning/papers/4","papers":[{"url":"/paper/learning-to-fly-via-deep-model-based","slug":"learning-to-fly-via-deep-model-based","title":"Learning to Fly via Deep Model-Based Reinforcement Learning","date":"2020-03-19","arxiv_id":"2003.08876","repositories_listed":1,"syntology":null},{"url":"/paper/fast-online-adaptation-in-robotics-through","slug":"fast-online-adaptation-in-robotics-through","title":"Fast Online Adaptation in Robotics through Meta-Learning Embeddings of Simulated Priors","date":"2020-03-10","arxiv_id":"2003.04663","repositories_listed":1,"syntology":null},{"url":"/paper/policy-aware-model-learning-for-policy","slug":"policy-aware-model-learning-for-policy","title":"Policy-Aware Model Learning for Policy Gradient Methods","date":"2020-02-28","arxiv_id":"2003.00030","repositories_listed":1,"syntology":null},{"url":"/paper/glib-exploration-via-goal-literal-babbling","slug":"glib-exploration-via-goal-literal-babbling","title":"GLIB: Efficient Exploration for Relational Model-Based Reinforcement Learning via Goal-Literal Babbling","date":"2020-01-22","arxiv_id":"2001.08299","repositories_listed":1,"syntology":null},{"url":"/paper/can-agents-learn-by-analogy-an-inferable","slug":"can-agents-learn-by-analogy-an-inferable","title":"Can Agents Learn by Analogy? An Inferable Model for PAC Reinforcement Learning","date":"2019-12-21","arxiv_id":"1912.10329","repositories_listed":1,"syntology":null},{"url":"/paper/a-model-based-reinforcement-learning-with-1","slug":"a-model-based-reinforcement-learning-with-1","title":"A Model-Based Reinforcement Learning with Adversarial Training for Online Recommendation","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/non-stationary-markov-decision-processes-a-1","slug":"non-stationary-markov-decision-processes-a-1","title":"Non-Stationary Markov Decision Processes, a Worst-Case Approach using Model-Based Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-model-based-reinforcement-learning-via","slug":"deep-model-based-reinforcement-learning-via","title":"Deep Model-Based Reinforcement Learning via Estimated Uncertainty and Conservative Policy Optimization","date":"2019-11-28","arxiv_id":"1911.12574","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-methods-for-model-based","slug":"asynchronous-methods-for-model-based","title":"Asynchronous Methods for Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12453","repositories_listed":1,"syntology":null},{"url":"/paper/entity-abstraction-in-visual-model-based","slug":"entity-abstraction-in-visual-model-based","title":"Entity Abstraction in Visual Model-Based Reinforcement Learning","date":"2019-10-28","arxiv_id":"1910.12827","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/entity-abstraction-in-visual-model-based#ran","syntology_url":"https://syntology.ai/paper/1910.12827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12827"}},"official":{"repos":["jcoreyes/OP3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-model-based-reinforcement-learning","slug":"towards-model-based-reinforcement-learning","title":"Towards Model-based Reinforcement Learning for Industry-near Environments","date":"2019-07-27","arxiv_id":"1907.11971","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-model-based-planning-with-policy","slug":"exploring-model-based-planning-with-policy","title":"Exploring Model-based Planning with Policy Networks","date":"2019-06-20","arxiv_id":"1906.08649","repositories_listed":1,"syntology":null},{"url":"/paper/calibrated-model-based-deep-reinforcement","slug":"calibrated-model-based-deep-reinforcement","title":"Calibrated Model-Based Deep Reinforcement Learning","date":"2019-06-19","arxiv_id":"1906.08312","repositories_listed":1,"syntology":null},{"url":"/paper/learning-powerful-policies-by-using","slug":"learning-powerful-policies-by-using","title":"Learning Powerful Policies by Using Consistent Dynamics Model","date":"2019-06-11","arxiv_id":"1906.04355","repositories_listed":1,"syntology":null},{"url":"/paper/extending-deep-model-predictive-control-with","slug":"extending-deep-model-predictive-control-with","title":"Safety Augmented Value Estimation from Demonstrations (SAVED): Safe Deep Model-Based RL for Sparse Cost Robotic Tasks","date":"2019-05-31","arxiv_id":"1905.13402","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extending-deep-model-predictive-control-with#ran","syntology_url":"https://syntology.ai/paper/1905.13402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.13402"}},"official":null}},{"url":"/paper/tight-regret-bounds-for-model-based","slug":"tight-regret-bounds-for-model-based","title":"Tight Regret Bounds for Model-Based Reinforcement Learning with Greedy Policies","date":"2019-05-27","arxiv_id":"1905.11527","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tight-regret-bounds-for-model-based#ran","syntology_url":"https://syntology.ai/paper/1905.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.11527"}},"official":{"repos":["NMerlis/TabulaRL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-residual-reinforcement-learning","slug":"deep-residual-reinforcement-learning","title":"Deep Residual Reinforcement Learning","date":"2019-05-03","arxiv_id":"1905.01072","repositories_listed":1,"syntology":null},{"url":"/paper/toyarchitecture-unsupervised-learning-of","slug":"toyarchitecture-unsupervised-learning-of","title":"ToyArchitecture: Unsupervised Learning of Interpretable Models of the World","date":"2019-03-20","arxiv_id":"1903.08772","repositories_listed":1,"syntology":null},{"url":"/paper/a-general-framework-for-structured-learning","slug":"a-general-framework-for-structured-learning","title":"A General Framework for Structured Learning of Mechanical Systems","date":"2019-02-22","arxiv_id":"1902.08705","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-general-framework-for-structured-learning#ran","syntology_url":"https://syntology.ai/paper/1902.08705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.08705"}},"official":{"repos":["sisl/mechamodlearn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-adversarial-user-model-for","slug":"generative-adversarial-user-model-for","title":"Generative Adversarial User Model for Reinforcement Learning Based Recommendation System","date":"2018-12-27","arxiv_id":"1812.10613","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/generative-adversarial-user-model-for#ran","syntology_url":"https://syntology.ai/paper/1812.10613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.10613"}},"official":{"repos":["xinshi-chen/GenerativeAdversarialUserModel"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/model-based-reinforcement-learning-via-meta","slug":"model-based-reinforcement-learning-via-meta","title":"Model-Based Reinforcement Learning via Meta-Policy Optimization","date":"2018-09-14","arxiv_id":"1809.05214","repositories_listed":1,"syntology":null},{"url":"/paper/solar-deep-structured-representations-for","slug":"solar-deep-structured-representations-for","title":"SOLAR: Deep Structured Representations for Model-Based Reinforcement Learning","date":"2018-08-28","arxiv_id":"1808.09105","repositories_listed":1,"syntology":null},{"url":"/paper/accurate-uncertainties-for-deep-learning","slug":"accurate-uncertainties-for-deep-learning","title":"Accurate Uncertainties for Deep Learning Using Calibrated Regression","date":"2018-07-01","arxiv_id":"1807.00263","repositories_listed":1,"syntology":null},{"url":"/paper/object-oriented-dynamics-predictor","slug":"object-oriented-dynamics-predictor","title":"Object-Oriented Dynamics Predictor","date":"2018-05-25","arxiv_id":"1806.07371","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-trainer-for-model-based","slug":"intelligent-trainer-for-model-based","title":"Intelligent Trainer for Model-Based Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09496","repositories_listed":1,"syntology":null},{"url":"/paper/lipschitz-continuity-in-model-based","slug":"lipschitz-continuity-in-model-based","title":"Lipschitz Continuity in Model-based Reinforcement Learning","date":"2018-04-19","arxiv_id":"1804.07193","repositories_listed":1,"syntology":null},{"url":"/paper/look-before-you-leap-bridging-model-free-and","slug":"look-before-you-leap-bridging-model-free-and","title":"Look Before You Leap: Bridging Model-Free and Model-Based Reinforcement Learning for Planned-Ahead Vision-and-Language Navigation","date":"2018-03-21","arxiv_id":"1803.07729","repositories_listed":1,"syntology":null},{"url":"/paper/learning-the-reward-function-for-a","slug":"learning-the-reward-function-for-a","title":"Learning the Reward Function for a Misspecified Model","date":"2018-01-29","arxiv_id":"1801.09624","repositories_listed":1,"syntology":null},{"url":"/paper/qlbs-q-learner-in-the-black-scholes-merton","slug":"qlbs-q-learner-in-the-black-scholes-merton","title":"QLBS: Q-Learner in the Black-Scholes(-Merton) Worlds","date":"2017-12-13","arxiv_id":"1712.04609","repositories_listed":1,"syntology":null},{"url":"/paper/learning-approximate-stochastic-transition","slug":"learning-approximate-stochastic-transition","title":"Learning Approximate Stochastic Transition Models","date":"2017-10-26","arxiv_id":"1710.09718","repositories_listed":1,"syntology":null},{"url":"/paper/safe-model-based-reinforcement-learning-with","slug":"safe-model-based-reinforcement-learning-with","title":"Safe Model-based Reinforcement Learning with Stability Guarantees","date":"2017-05-23","arxiv_id":"1705.08551","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multimodal-transition-dynamics-for","slug":"learning-multimodal-transition-dynamics-for","title":"Learning Multimodal Transition Dynamics for Model-Based Reinforcement Learning","date":"2017-05-01","arxiv_id":"1705.00470","repositories_listed":1,"syntology":null},{"url":"/paper/self-correcting-models-for-model-based","slug":"self-correcting-models-for-model-based","title":"Self-Correcting Models for Model-Based Reinforcement Learning","date":"2016-12-19","arxiv_id":"1612.06018","repositories_listed":1,"syntology":null},{"url":"/paper/deep-visual-foresight-for-planning-robot","slug":"deep-visual-foresight-for-planning-robot","title":"Deep Visual Foresight for Planning Robot Motion","date":"2016-10-03","arxiv_id":"1610.00696","repositories_listed":1,"syntology":null},{"url":null,"slug":"on-quantum-bsde-solver-for-high-dimensional","title":"On Quantum BSDE Solver for High-Dimensional Parabolic PDEs","date":"2025-06-17","arxiv_id":"2506.14612","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-entropy-regularized-reinforcement","title":"Relative Entropy Regularized Reinforcement Learning for Efficient Encrypted Policy Synthesis","date":"2025-06-14","arxiv_id":"2506.12358","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-model-based-reinforcement-1","title":"Accelerating Model-Based Reinforcement Learning using Non-Linear Trajectory Optimization","date":"2025-06-03","arxiv_id":"2506.02767","repositories_listed":0,"syntology":null},{"url":null,"slug":"bregman-centroid-guided-cross-entropy-method","title":"Bregman Centroid Guided Cross-Entropy Method","date":"2025-06-02","arxiv_id":"2506.02205","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-models-for-cognitive-agents","title":"World Models for Cognitive Agents: Transforming Edge Intelligence in Future Networks","date":"2025-05-31","arxiv_id":"2506.00417","repositories_listed":0,"syntology":null},{"url":null,"slug":"calibrated-value-aware-model-learning-with","title":"Calibrated Value-Aware Model Learning with Stochastic Environment Models","date":"2025-05-28","arxiv_id":"2505.22772","repositories_listed":0,"syntology":null},{"url":null,"slug":"jedi-latent-end-to-end-diffusion-mitigates","title":"JEDI: Latent End-to-end Diffusion Mitigates Agent-Human Performance Asymmetry in Model-Based Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19698","repositories_listed":0,"syntology":null},{"url":null,"slug":"meddreamer-model-based-reinforcement-learning","title":"MedDreamer: Model-Based Reinforcement Learning with Latent Imagination on Complex EHRs for Clinical Decision Support","date":"2025-05-26","arxiv_id":"2505.19785","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaze-into-the-abyss-planning-to-seek-entropy","title":"Gaze Into the Abyss -- Planning to Seek Entropy When Reward is Scarce","date":"2025-05-22","arxiv_id":"2505.16787","repositories_listed":0,"syntology":null},{"url":"/paper/raw2drive-reinforcement-learning-with-aligned","slug":"raw2drive-reinforcement-learning-with-aligned","title":"Raw2Drive: Reinforcement Learning with Aligned World Models for End-to-End Autonomous Driving (in CARLA v2)","date":"2025-05-22","arxiv_id":"2505.16394","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-planning-and-mbrl-with-temporally","title":"Improving planning and MBRL with temporally-extended actions","date":"2025-05-21","arxiv_id":"2505.15754","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-driven-world-model-adaptation-for","title":"Policy-Driven World Model Adaptation for Robust Offline Model-based Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.13709","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-distance-aware-transition","title":"Temporal Distance-aware Transition Augmentation for Offline Model-based Reinforcement Learning","date":"2025-05-19","arxiv_id":"2505.13144","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-goal-dexterous-hand-manipulation-using","title":"Multi-Goal Dexterous Hand Manipulation using Probabilistic Model-based Reinforcement Learning","date":"2025-04-30","arxiv_id":"2504.21585","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-assimilated-model-based-reinforcement","title":"Data-Assimilated Model-Based Reinforcement Learning for Partially Observed Chaotic Flows","date":"2025-04-23","arxiv_id":"2504.16588","repositories_listed":0,"syntology":null},{"url":null,"slug":"pin-wm-learning-physics-informed-world-models","title":"PIN-WM: Learning Physics-INformed World Models for Non-Prehensile Manipulation","date":"2025-04-23","arxiv_id":"2504.16693","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-global-control-of-underactuated","title":"Learning global control of underactuated systems with Model-Based Reinforcement Learning","date":"2025-04-09","arxiv_id":"2504.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-imperfect-models-when-multi","title":"Learning with Imperfect Models: When Multi-step Prediction Mitigates Compounding Error","date":"2025-04-02","arxiv_id":"2504.01766","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-before-leap-look-ahead-planning-with","title":"Look Before Leap: Look-Ahead Planning with Uncertainty in Reinforcement Learning","date":"2025-03-26","arxiv_id":"2503.20139","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-regularized-gradient-estimators-for","title":"Entropy-regularized Gradient Estimators for Approximate Bayesian Inference","date":"2025-03-15","arxiv_id":"2503.11964","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-causal-model-based-policy","title":"Towards Causal Model-Based Policy Optimization","date":"2025-03-12","arxiv_id":"2503.09719","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-traffic-signal-control-through","title":"Enhancing Traffic Signal Control through Model-based Reinforcement Learning and Policy Reuse","date":"2025-03-11","arxiv_id":"2503.08728","repositories_listed":0,"syntology":null},{"url":null,"slug":"indrive-intrinsic-disagreement-based","title":"InDRiVE: Intrinsic Disagreement based Reinforcement for Vehicle Exploration through Curiosity Driven Generalized World Model","date":"2025-03-07","arxiv_id":"2503.05573","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-retention-for-continual-model-based","title":"Knowledge Retention for Continual Model-Based Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04256","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-transformer-based-world-models-with","title":"Learning Transformer-based World Models with Contrastive Predictive Coding","date":"2025-03-06","arxiv_id":"2503.04416","repositories_listed":0,"syntology":null},{"url":null,"slug":"world-models-for-anomaly-detection-during","title":"World Models for Anomaly Detection during Model-Based Reinforcement Learning Inference","date":"2025-03-04","arxiv_id":"2503.02552","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-information-enhanced-model","title":"Differentiable Information Enhanced Model-Based Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-dreaming-a-global-workspace","title":"Multimodal Dreaming: A Global Workspace Approach to World Model-Based Reinforcement Learning","date":"2025-02-28","arxiv_id":"2502.21142","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-model-based-reinforcement","title":"Accelerating Model-Based Reinforcement Learning with State-Space World Models","date":"2025-02-27","arxiv_id":"2502.20168","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-offline-model-based-rl-via-active","title":"Enhancing Offline Model-Based RL via Active Model Selection: A Bayesian Optimization Perspective","date":"2025-02-17","arxiv_id":"2502.11480","repositories_listed":0,"syntology":null},{"url":null,"slug":"privilegeddreamer-explicit-imagination-of","title":"PrivilegedDreamer: Explicit Imagination of Privileged Information for Rapid Adaptation of Learned Policies","date":"2025-02-17","arxiv_id":"2502.11377","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-empowerment-gain-through-causal","title":"Towards Empowerment Gain through Causal Structure Learning in Model-Based RL","date":"2025-02-14","arxiv_id":"2502.10077","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-trained-video-generative-models-as-world","title":"Pre-Trained Video Generative Models as World Simulators","date":"2025-02-10","arxiv_id":"2502.07825","repositories_listed":0,"syntology":null},{"url":null,"slug":"td-m-pc-2-improving-temporal-difference-mpc","title":"TD-M(PC)$^2$: Improving Temporal Difference MPC Through Policy Constraint","date":"2025-02-05","arxiv_id":"2502.03550","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ai-for-lyapunov-optimization","title":"Generative AI for Lyapunov Optimization Theory in UAV-based Low-Altitude Economy Networking","date":"2025-01-27","arxiv_id":"2501.15928","repositories_listed":0,"syntology":null},{"url":null,"slug":"objects-matter-object-centric-world-models","title":"Objects matter: object-centric world models improve reinforcement learning in visually complex environments","date":"2025-01-27","arxiv_id":"2501.16443","repositories_listed":0,"syntology":null},{"url":null,"slug":"adawm-adaptive-world-model-based-planning-for","title":"AdaWM: Adaptive World Model based Planning for Autonomous Driving","date":"2025-01-22","arxiv_id":"2501.13072","repositories_listed":0,"syntology":null},{"url":null,"slug":"glam-global-local-variation-awareness-in","title":"GLAM: Global-Local Variation Awareness in Mamba-based World Model","date":"2025-01-21","arxiv_id":"2501.11949","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-world-model-a-neural-network","title":"Robotic World Model: A Neural Network Simulator for Robust Policy Optimization in Robotics","date":"2025-01-17","arxiv_id":"2501.10100","repositories_listed":0,"syntology":null},{"url":null,"slug":"evade-event-based-variational-thompson-1","title":"EVaDE : Event-Based Variational Thompson Sampling for Model-Based Reinforcement Learning","date":"2025-01-16","arxiv_id":"2501.09611","repositories_listed":0,"syntology":null},{"url":null,"slug":"computing-approximated-fixpoints-via-dampened","title":"Approximating Fixpoints of Approximated Functions","date":"2025-01-15","arxiv_id":"2501.08950","repositories_listed":0,"syntology":null},{"url":null,"slug":"inferring-transition-dynamics-from-value","title":"Inferring Transition Dynamics from Value Functions","date":"2025-01-15","arxiv_id":"2501.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reduced-order-iterative-linear-quadratic","title":"A Reduced Order Iterative Linear Quadratic Regulator (ILQR) Technique for the Optimal Control of Nonlinear Partial Differential Equations","date":"2025-01-11","arxiv_id":"2501.06635","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-transfer-in-model-based","title":"Knowledge Transfer in Model-Based Reinforcement Learning Agents for Efficient Multi-Task Learning","date":"2025-01-09","arxiv_id":"2501.05329","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-calibration-with-model-based","title":"Digital Twin Calibration with Model-Based Reinforcement Learning","date":"2025-01-04","arxiv_id":"2501.02205","repositories_listed":0,"syntology":null},{"url":"/paper/stealing-that-free-lunch-exposing-the-limits","slug":"stealing-that-free-lunch-exposing-the-limits","title":"Stealing That Free Lunch: Exposing the Limits of Dyna-Style Reinforcement Learning","date":"2024-12-18","arxiv_id":"2412.14312","repositories_listed":0,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stealing-that-free-lunch-exposing-the-limits#ran","syntology_url":"https://syntology.ai/paper/2412.14312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14312"}},"official":null}},{"url":null,"slug":"learning-to-navigate-in-mazes-with-novel","title":"Learning to Navigate in Mazes with Novel Layouts using Abstract Top-down Maps","date":"2024-12-16","arxiv_id":"2412.12024","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-shaped-prediction-avoiding","title":"Policy-shaped prediction: avoiding distractions in model-based reinforcement learning","date":"2024-12-08","arxiv_id":"2412.05766","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-optimal-control-of-unknown","title":"Data-driven optimal control of unknown nonlinear dynamical systems using the Koopman operator","date":"2024-12-02","arxiv_id":"2412.01085","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-learning-with-temporal-gaussian","title":"Structure learning with Temporal Gaussian Mixture for model-based Reinforcement Learning","date":"2024-11-18","arxiv_id":"2411.11511","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagine-2-drive-high-fidelity-world-modeling","title":"Imagine-2-Drive: Leveraging High-Fidelity World Models via Multi-Modal Diffusion Policies","date":"2024-11-15","arxiv_id":"2411.10171","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-surprising-ineffectiveness-of-pre-trained","title":"The Surprising Ineffectiveness of Pre-Trained Visual Representations for Model-Based Reinforcement Learning","date":"2024-11-15","arxiv_id":"2411.10175","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-transition-learning-learning-dynamics","title":"Inverse Transition Learning: Learning Dynamics from Demonstrations","date":"2024-11-07","arxiv_id":"2411.05174","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-to-trust-your-data-enhancing-dyna-style","title":"When to Trust Your Data: Enhancing Dyna-Style Model-Based Reinforcement Learning With Data Filter","date":"2024-10-16","arxiv_id":"2410.12160","repositories_listed":0,"syntology":null},{"url":null,"slug":"dodt-enhanced-online-decision-transformer","title":"DODT: Enhanced Online Decision Transformer Learning through Dreamer's Actor-Critic Trajectory Forecasting","date":"2024-10-15","arxiv_id":"2410.11359","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-the-pertinent-salient-task-relevant","title":"Make the Pertinent Salient: Task-Relevant Reconstruction for Visual Control with Distractions","date":"2024-10-13","arxiv_id":"2410.09972","repositories_listed":0,"syntology":null},{"url":null,"slug":"sold-reinforcement-learning-with-slot-object","title":"SOLD: Slot Object-Centric Latent Dynamics Models for Relational Manipulation Learning from Pixels","date":"2024-10-11","arxiv_id":"2410.08822","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-based-reinforcement-learning-2","title":"Efficient Model-Based Reinforcement Learning Through Optimistic Thompson Sampling","date":"2024-10-07","arxiv_id":"2410.04988","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-versus-model-based-reinforcement","title":"Model-Free versus Model-Based Reinforcement Learning for Fixed-Wing UAV Attitude Control Under Varying Wind Conditions","date":"2024-09-26","arxiv_id":"2409.17896","repositories_listed":0,"syntology":null},{"url":null,"slug":"colanet-a-spiking-neural-network-with","title":"CoLaNET -- A Spiking Neural Network with Columnar Layered Architecture for Classification","date":"2024-09-02","arxiv_id":"2409.01230","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-analogical-reasoning-in-the","title":"Enhancing Analogical Reasoning in the Abstraction and Reasoning Corpus via Model-Based RL","date":"2024-08-27","arxiv_id":"2408.14855","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-9","title":"Model-Based Reinforcement Learning for Control of Strongly-Disturbed Unsteady Aerodynamic Flows","date":"2024-08-26","arxiv_id":"2408.14685","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-model-based-reinforcement-learning","title":"Offline Model-Based Reinforcement Learning with Anti-Exploration","date":"2024-08-20","arxiv_id":"2408.10713","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-rl-as-a-minimalist-approach-to","title":"Model-based RL as a Minimalist Approach to Horizon-Free and Second-Order Bounds","date":"2024-08-16","arxiv_id":"2408.08994","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-interpretability-of-codebooks-in-model","title":"The Interpretability of Codebooks in Model-Based Reinforcement Learning is Limited","date":"2024-07-28","arxiv_id":"2407.19532","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-flexible-behaviour-by-the","title":"Modeling flexible behavior with remapping-based hippocampal sequence learning","date":"2024-07-20","arxiv_id":"2407.14708","repositories_listed":0,"syntology":null}],"record_sha256":"05185f0173b915116d4a957cfadf928b9a099d7066e9718edc6e9d2539ae3bb1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}