{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/model-based-reinforcement-learning/papers/4","list_of":"/task/model-based-reinforcement-learning","task":"Model-based Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":8,"rows_per_page":100,"rows":[301,400],"of":708,"counts":{"archive_papers_tagged":708,"with_a_code_link":234,"where_syntology_ran_a_sample":72,"not_listed_spam_title":0,"listed":708,"listed_where_code_ran":72,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":66,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":66,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/model-based-reinforcement-learning","prev":"/task/model-based-reinforcement-learning/papers/3","next":"/task/model-based-reinforcement-learning/papers/5","papers":[{"url":null,"slug":"model-based-policy-optimization-using","title":"Model-based Policy Optimization using Symbolic World Model","date":"2024-07-18","arxiv_id":"2407.13518","repositories_listed":0,"syntology":null},{"url":null,"slug":"because-bilinear-causal-representation-for","title":"BECAUSE: Bilinear Causal Representation for Generalizable Offline Model-based Reinforcement Learning","date":"2024-07-15","arxiv_id":"2407.10967","repositories_listed":0,"syntology":null},{"url":null,"slug":"gnn-with-model-based-rl-for-multi-agent","title":"Graph Neural Networks with Model-based Reinforcement Learning for Multi-agent Systems","date":"2024-07-12","arxiv_id":"2407.09249","repositories_listed":0,"syntology":null},{"url":null,"slug":"fosp-fine-tuning-offline-safe-policy-through","title":"FOSP: Fine-tuning Offline Safe Policy through World Models","date":"2024-07-06","arxiv_id":"2407.04942","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-rag-empowered-multi-modal-llm-for","title":"Hybrid RAG-empowered Multi-modal LLM for Secure Data Management in Internet of Medical Things: A Diffusion-based Contract Approach","date":"2024-07-01","arxiv_id":"2407.00978","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-ordinary-differential-equations","title":"Identifying Ordinary Differential Equations for Data-efficient Model-based Reinforcement Learning","date":"2024-06-28","arxiv_id":"2406.19817","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-and-input-constrained-output-feedback","title":"State and Input Constrained Output-Feedback Adaptive Optimal Control of Affine Nonlinear Systems","date":"2024-06-27","arxiv_id":"2406.18804","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-infinite-horizon","title":"Infinite-Horizon Reinforcement Learning with Multinomial Logistic Function Approximation","date":"2024-06-19","arxiv_id":"2406.13633","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-architecture","title":"Reinforcement learning-based architecture search for quantum machine learning","date":"2024-06-04","arxiv_id":"2406.02717","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-view-on-planning-in-online","title":"A New View on Planning in Online Reinforcement Learning","date":"2024-06-03","arxiv_id":"2406.01562","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-layer-splitting-for-wireless-llm","title":"Adaptive Layer Splitting for Wireless LLM Inference in Edge Computing: A Model-Based Reinforcement Learning Approach","date":"2024-06-03","arxiv_id":"2406.02616","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-play-atari-in-a-world-of-tokens","title":"Learning to Play Atari in a World of Tokens","date":"2024-06-03","arxiv_id":"2406.01361","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-horizon-actor-critic-for-policy","title":"Adaptive Horizon Actor-Critic for Policy Learning in Contact-Rich Differentiable Simulation","date":"2024-05-28","arxiv_id":"2405.17784","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-models-for-building-adaptive-model","title":"Partial Models for Building Adaptive Model-Based Reinforcement Learning Agents","date":"2024-05-27","arxiv_id":"2405.16899","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-8","title":"Model-based reinforcement learning for protein backbone design","date":"2024-05-03","arxiv_id":"2405.01983","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-model-based-reinforcement-learning-1","title":"Continual Model-based Reinforcement Learning for Data Efficient Wireless Network Optimisation","date":"2024-04-30","arxiv_id":"2404.19462","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-note-on-loss-functions-and-error","title":"A Note on Loss Functions and Error Compounding in Model-based Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.09946","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-for-control-oriented","title":"Active Learning for Control-Oriented Identification of Nonlinear Systems","date":"2024-04-13","arxiv_id":"2404.09030","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-exploration-in-bayesian-model-based","title":"Active Exploration in Bayesian Model-based Reinforcement Learning for Robot Manipulation","date":"2024-04-02","arxiv_id":"2404.01867","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-model-based-reinforcement-learning","title":"Robust Model Based Reinforcement Learning Using $\\mathcal{L}_1$ Adaptive Control","date":"2024-03-21","arxiv_id":"2403.14860","repositories_listed":0,"syntology":null},{"url":null,"slug":"dr-strategy-model-based-generalist-agents","title":"Dr. Strategy: Model-Based Generalist Agents with Strategic Dreaming","date":"2024-02-29","arxiv_id":"2402.18866","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-in-doubt-think-slow-iterative-reasoning","title":"When in Doubt, Think Slow: Iterative Reasoning with Latent Imagination","date":"2024-02-23","arxiv_id":"2402.15283","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-control-of","title":"Model-Based Reinforcement Learning Control of Reaction-Diffusion Problems","date":"2024-02-22","arxiv_id":"2402.14446","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-model-based-reinforcement","title":"Towards Robust Model-Based Reinforcement Learning Against Adversarial Corruption","date":"2024-02-14","arxiv_id":"2402.08991","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-model-based-offline","title":"Differentially Private Deep Model-Based Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-step-loss-function-for-robust","title":"A Multi-step Loss Function for Robust Learning of the Dynamics in Model-based Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.03146","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-autoregressive-density-nets-vs-neural","title":"Deep autoregressive density nets vs neural ensembles for model-based offline reinforcement learning","date":"2024-02-05","arxiv_id":"2402.02858","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-in-stochastic-environment-with-delays","title":"Control in Stochastic Environment with Delays: A Model-based Reinforcement Learning Approach","date":"2024-02-01","arxiv_id":"2402.00313","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduled-curiosity-deep-dyna-q-efficient","title":"Scheduled Curiosity-Deep Dyna-Q: Efficient Exploration for Dialog Policy Learning","date":"2024-01-31","arxiv_id":"2402.00085","repositories_listed":0,"syntology":null},{"url":null,"slug":"locality-sensitive-sparse-encoding-for","title":"Locality Sensitive Sparse Encoding for Learning World Models Online","date":"2024-01-23","arxiv_id":"2401.13034","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-and-causality-in-complex-systems-a","title":"Emergence and Causality in Complex Systems: A Survey on Causal Emergence and Related Quantitative Studies","date":"2023-12-28","arxiv_id":"2312.16815","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-programming-based-approximate-optimal","title":"Dynamic Programming-based Approximate Optimal Control for Model-Based Reinforcement Learning","date":"2023-12-22","arxiv_id":"2312.14463","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-epistemic-variance-of-values-for","title":"Model-Based Epistemic Variance of Values for Risk-Aware Policy Optimization","date":"2023-12-07","arxiv_id":"2312.04386","repositories_listed":0,"syntology":null},{"url":null,"slug":"score-aware-policy-gradient-methods-and","title":"Score-Aware Policy-Gradient Methods and Performance Guarantees using Local Lyapunov Conditions: Applications to Product-Form Stochastic Networks and Queueing Systems","date":"2023-12-05","arxiv_id":"2312.02804","repositories_listed":0,"syntology":null},{"url":"/paper/regularity-as-intrinsic-reward-for-free-play-1","slug":"regularity-as-intrinsic-reward-for-free-play-1","title":"Regularity as Intrinsic Reward for Free Play","date":"2023-12-03","arxiv_id":"2312.01473","repositories_listed":0,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":16,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/regularity-as-intrinsic-reward-for-free-play-1#ran","syntology_url":"https://syntology.ai/paper/2312.01473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01473"}},"official":null}},{"url":null,"slug":"langwm-language-grounded-world-model","title":"LanGWM: Language Grounded World Model","date":"2023-11-29","arxiv_id":"2311.17593","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-port-navigation-with-ranging","title":"Autonomous Port Navigation With Ranging Sensors Using Model-Based Reinforcement Learning","date":"2023-11-17","arxiv_id":"2312.05257","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-introduction-to-reinforcement-learning-for","title":"An introduction to reinforcement learning for neuroscience","date":"2023-11-13","arxiv_id":"2311.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-twinning-from-digital-twins-to","title":"Reinforcement Twinning: from digital twins to model-based reinforcement learning","date":"2023-11-07","arxiv_id":"2311.03628","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreamsmooth-improving-model-based","title":"DreamSmooth: Improving Model-based Reinforcement Learning via Reward Smoothing","date":"2023-11-02","arxiv_id":"2311.01450","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-alignment-ceiling-objective-mismatch-in","title":"The Alignment Ceiling: Objective Mismatch in Reinforcement Learning from Human Feedback","date":"2023-10-31","arxiv_id":"2311.00168","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-in-continuous-time","title":"Efficient Exploration in Continuous-time Model-based Reinforcement Learning","date":"2023-10-30","arxiv_id":"2310.19848","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphical-object-centric-actor-critic","title":"Relational Object-Centric Actor-Critic","date":"2023-10-26","arxiv_id":"2310.17178","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-primacy-bias-in-model-based-rl","title":"Mind the Model, Not the Agent: The Primacy Bias in Model-based RL","date":"2023-10-23","arxiv_id":"2310.15017","repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-search-in-dag-space-with-model-based","title":"Tree Search in DAG Space with Model-based Reinforcement Learning for Causal Discovery","date":"2023-10-20","arxiv_id":"2310.13576","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-biased-maximum-likelihood-estimation","title":"Value-Biased Maximum Likelihood Estimation for Model-based Reinforcement Learning in Discounted Linear MDPs","date":"2023-10-17","arxiv_id":"2310.11515","repositories_listed":0,"syntology":null},{"url":null,"slug":"moconvq-unified-physics-based-motion-control","title":"MoConVQ: Unified Physics-Based Motion Control via Scalable Discrete Representations","date":"2023-10-16","arxiv_id":"2310.10198","repositories_listed":0,"syntology":null},{"url":null,"slug":"coplanner-plan-to-roll-out-conservatively-but","title":"COPlanner: Plan to Roll Out Conservatively but to Explore Optimistically for Model-Based RL","date":"2023-10-11","arxiv_id":"2310.07220","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-timestep-models-for-model-based","title":"Multi-timestep models for Model-based Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05672","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-consistent-dynamics-models-are","title":"Reward-Consistent Dynamics Models are Strongly Generalizable for Offline Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05422","repositories_listed":0,"syntology":null},{"url":null,"slug":"amortized-network-intervention-to-steer-the","title":"Amortized Network Intervention to Steer the Excitatory Point Processes","date":"2023-10-06","arxiv_id":"2310.04159","repositories_listed":0,"syntology":null},{"url":null,"slug":"modem-v2-visuo-motor-world-models-for-real","title":"MoDem-V2: Visuo-Motor World Models for Real-World Robot Manipulation","date":"2023-09-25","arxiv_id":"2309.14236","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-mildly-conservative-model-based","title":"DOMAIN: MilDly COnservative Model-BAsed OfflINe Reinforcement Learning","date":"2023-09-16","arxiv_id":"2309.08925","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-the-uncertainty-risk-aware-and-actively","title":"Mind the Uncertainty: Risk-Aware and Actively Exploring Model-Based Reinforcement Learning","date":"2023-09-11","arxiv_id":"2309.05582","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributionally-robust-model-based","title":"Distributionally Robust Model-based Reinforcement Learning with Large State Spaces","date":"2023-09-05","arxiv_id":"2309.02236","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-potential-of-world-models-for","title":"Exploring the Potential of World Models for Anomaly Detection in Autonomous Driving","date":"2023-08-10","arxiv_id":"2308.05701","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretically-guaranteed-policy-improvement","title":"Theoretically Guaranteed Policy Improvement Distilled from Model-Based Planning","date":"2023-07-24","arxiv_id":"2307.12933","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-transformation-sequence-retrieval-with","title":"Image Transformation Sequence Retrieval with General Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.06630","repositories_listed":0,"syntology":null},{"url":null,"slug":"facing-off-world-model-backbones-rnns","title":"Facing Off World Model Backbones: RNNs, Transformers, and S4","date":"2023-07-05","arxiv_id":"2307.02064","repositories_listed":0,"syntology":null},{"url":null,"slug":"surge-routing-event-informed-multiagent","title":"Surge Routing: Event-informed Multiagent Reinforcement Learning for Autonomous Rideshare","date":"2023-07-05","arxiv_id":"2307.02637","repositories_listed":0,"syntology":null},{"url":null,"slug":"l-ac-learning-latent-decision-aware-models","title":"$λ$-models: Effective Decision-Aware Reinforcement Learning with Latent Models","date":"2023-06-30","arxiv_id":"2306.17366","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-generative-models-for-decision-making","title":"Deep Generative Models for Decision-Making and Control","date":"2023-06-15","arxiv_id":"2306.08810","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-learn-and-generalize-from-three","title":"How to Learn and Generalize From Three Minutes of Data: Physics-Constrained and Uncertainty-Aware Neural Stochastic Differential Equations","date":"2023-06-10","arxiv_id":"2306.06335","repositories_listed":0,"syntology":null},{"url":null,"slug":"iql-td-mpc-implicit-q-learning-for","title":"IQL-TD-MPC: Implicit Q-Learning for Hierarchical Model Predictive Control","date":"2023-06-01","arxiv_id":"2306.00867","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-model-does-muzero-learn","title":"What model does MuZero learn?","date":"2023-06-01","arxiv_id":"2306.00840","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-based-3d-map-management-for-edge","title":"Digital Twin-Based 3D Map Management for Edge-Assisted Mobile Augmented Reality","date":"2023-05-26","arxiv_id":"2305.16571","repositories_listed":0,"syntology":null},{"url":null,"slug":"tom-learning-policy-aware-models-for-model","title":"TOM: Learning Policy-Aware Models for Model-Based Reinforcement Learning via Transition Occupancy Matching","date":"2023-05-22","arxiv_id":"2305.12663","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-active-exploration-and-uncertainty","title":"Bridging Active Exploration and Uncertainty-Aware Deployment Using Probabilistic Ensemble Neural Network Dynamics","date":"2023-05-20","arxiv_id":"2305.12240","repositories_listed":0,"syntology":null},{"url":null,"slug":"sense-imagine-act-multimodal-perception","title":"Sense, Imagine, Act: Multimodal Perception Improves Model-Based Reinforcement Learning for Head-to-Head Autonomous Racing","date":"2023-05-08","arxiv_id":"2305.04750","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-offline-model-based-reinforcement","title":"A Survey on Offline Model-Based Reinforcement Learning","date":"2023-05-05","arxiv_id":"2305.03360","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-machine-co-adaption-interface-via","title":"Human Machine Co-adaption Interface via Cooperation Markov Decision Process System","date":"2023-05-03","arxiv_id":"2305.02058","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-6","title":"Model Based Reinforcement Learning for Personalized Heparin Dosing","date":"2023-04-19","arxiv_id":"2304.10000","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-decision-focused-learning-for-reward","title":"Decision-Focused Model-based Reinforcement Learning for Reward Transfer","date":"2023-04-06","arxiv_id":"2304.03365","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-and-parameter-estimation-for-affine","title":"State and Parameter Estimation for Affine Nonlinear Systems","date":"2023-04-04","arxiv_id":"2304.01526","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-and-robust-model-based","title":"Risk-Sensitive and Robust Model-Based Reinforcement Learning and Planning","date":"2023-04-02","arxiv_id":"2304.00573","repositories_listed":0,"syntology":null},{"url":null,"slug":"edgi-equivariant-diffusion-for-planning-with","title":"EDGI: Equivariant Diffusion for Planning with Embodied Agents","date":"2023-03-22","arxiv_id":"2303.12410","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-policy-iteration-algorithm-for","title":"A New Policy Iteration Algorithm For Reinforcement Learning in Zero-Sum Markov Games","date":"2023-03-17","arxiv_id":"2303.09716","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-benefits-of-leveraging-structural","title":"On the Benefits of Leveraging Structural Information in Planning Over the Learned Model","date":"2023-03-15","arxiv_id":"2303.08856","repositories_listed":0,"syntology":null},{"url":null,"slug":"replay-buffer-with-local-forgetting-for","title":"Replay Buffer with Local Forgetting for Adapting to Local Environment Changes in Deep Model-Based Reinforcement Learning","date":"2023-03-15","arxiv_id":"2303.08690","repositories_listed":0,"syntology":null},{"url":null,"slug":"beware-of-instantaneous-dependence-in","title":"Beware of Instantaneous Dependence in Reinforcement Learning","date":"2023-03-09","arxiv_id":"2303.05458","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximating-energy-market-clearing-and","title":"Approximating Energy Market Clearing and Bidding With Model-Based Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01772","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-effect-of-varying-amounts","title":"Understanding the effect of varying amounts of replay per step","date":"2023-02-20","arxiv_id":"2302.10311","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-muzero","title":"Equivariant MuZero","date":"2023-02-09","arxiv_id":"2302.04798","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-model-ensemble-necessary-model-based-rl","title":"Is Model Ensemble Necessary? Model-based RL via a Single Model with Lipschitz Regularized Value Function","date":"2023-02-02","arxiv_id":"2302.01244","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-control-from-raw-position","title":"Learning Control from Raw Position Measurements","date":"2023-01-30","arxiv_id":"2301.13183","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-stein-information-directed","title":"STEERING: Stein Information Directed Exploration for Model-Based Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12038","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsic-motivation-in-model-based","title":"Intrinsic Motivation in Model-based Reinforcement Learning: A Brief Review","date":"2023-01-24","arxiv_id":"2301.10067","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimal-value-equivalent-partial-models-for","title":"Minimal Value-Equivalent Partial Models for Scalable and Robust Planning in Lifelong Reinforcement Learning","date":"2023-01-24","arxiv_id":"2301.10119","repositories_listed":0,"syntology":null},{"url":null,"slug":"stock-trading-optimization-through-model-1","title":"Model Based Reinforcement Learning with Non-Gaussian Environment Dynamics and its Application to Portfolio Optimization","date":"2023-01-23","arxiv_id":"2301.09297","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-in-model-based-reinforcement-1","title":"Exploration in Model-based Reinforcement Learning with Randomized Reward","date":"2023-01-09","arxiv_id":"2301.03142","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-with-2","title":"Model-Based Reinforcement Learning with Multinomial Logistic Function Approximation","date":"2022-12-27","arxiv_id":"2212.13540","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-variable-representation-for","title":"Latent Variable Representation for Reinforcement Learning","date":"2022-12-17","arxiv_id":"2212.08765","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-data-driven-pricing-scheme-for-optimal","title":"A Data-driven Pricing Scheme for Optimal Routing through Artificial Currencies","date":"2022-11-27","arxiv_id":"2211.14793","repositories_listed":0,"syntology":null},{"url":null,"slug":"prototypical-context-aware-dynamics","title":"Prototypical context-aware dynamics generalization for high-dimensional model-based reinforcement learning","date":"2022-11-23","arxiv_id":"2211.12774","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-estimation-of-controlled-markov","title":"Offline Estimation of Controlled Markov Chains: Minimaxity and Sample Complexity","date":"2022-11-14","arxiv_id":"2211.07092","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-value-learning-implicit-models","title":"Contrastive Value Learning: Implicit Models for Simple Offline RL","date":"2022-11-03","arxiv_id":"2211.02100","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-with-a-1","title":"Model-based Reinforcement Learning with a Hamiltonian Canonical ODE Network","date":"2022-11-02","arxiv_id":"2211.00942","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam-rl-sensing-aware-model-based","title":"SAM-RL: Sensing-Aware Model-Based Reinforcement Learning via Differentiable Physics-Based Simulation and Rendering","date":"2022-10-27","arxiv_id":"2210.15185","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-exploration-for-robotic-manipulation","title":"Active Exploration for Robotic Manipulation","date":"2022-10-23","arxiv_id":"2210.12806","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-uncertainty-deep-exploration-in","title":"Epistemic Monte Carlo Tree Search","date":"2022-10-21","arxiv_id":"2210.13455","repositories_listed":0,"syntology":null}],"record_sha256":"c9595a1854551e6e85b92e971e983daf8bc4a1ee49e730289cc9a058e3ad85fe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}