{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/model-based-reinforcement-learning/papers/5","list_of":"/task/model-based-reinforcement-learning","task":"Model-based Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":8,"rows_per_page":100,"rows":[401,500],"of":708,"counts":{"archive_papers_tagged":708,"with_a_code_link":234,"where_syntology_ran_a_sample":72,"not_listed_spam_title":0,"listed":708,"listed_where_code_ran":72,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":66,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":66,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/model-based-reinforcement-learning","prev":"/task/model-based-reinforcement-learning/papers/4","next":"/task/model-based-reinforcement-learning/papers/6","papers":[{"url":null,"slug":"output-feedback-adaptive-optimal-control-of","title":"Output Feedback Adaptive Optimal Control of Affine Nonlinear systems with a Linear Measurement Model","date":"2022-10-13","arxiv_id":"2210.06637","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-inventory-management","title":"Deep Inventory Management","date":"2022-10-06","arxiv_id":"2210.03137","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-unsupervised-learning-of-world","title":"Contrastive Unsupervised Learning of World Model with Invariant Causal Features","date":"2022-09-29","arxiv_id":"2209.14932","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-parsimonious-dynamics-for","title":"Learning Parsimonious Dynamics for Generalization in Reinforcement Learning","date":"2022-09-29","arxiv_id":"2209.14781","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-mle-a-generic-model-based","title":"Optimistic MLE -- A Generic Model-based Algorithm for Partially Observable Sequential Decision Making","date":"2022-09-29","arxiv_id":"2209.14997","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-non-exponential","title":"Reinforcement Learning with Non-Exponential Discounting","date":"2022-09-27","arxiv_id":"2209.13413","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-optimization-for-variable","title":"Performance Optimization for Variable Bitwidth Federated Learning in Wireless Networks","date":"2022-09-21","arxiv_id":"2209.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-spiking-neural-network-learning-markov","title":"A Spiking Neural Network Learning Markov Chain","date":"2022-09-20","arxiv_id":"2209.09572","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-dual-policy-optimization-for","title":"Conservative Dual Policy Optimization for Efficient Model-Based Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.07676","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-summation-a-novel-scoring-function-for","title":"Value Summation: A Novel Scoring Function for MPC-based Model-based Reinforcement Learning","date":"2022-09-16","arxiv_id":"2209.08169","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-modulated-model-based-offline","title":"Concept-modulated model-based offline reinforcement learning for rapid generalization","date":"2022-09-07","arxiv_id":"2209.03207","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-inference-for-model-free-and","title":"Variational Inference for Model-Free and Model-Based Reinforcement Learning","date":"2022-09-04","arxiv_id":"2209.01693","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-abstracted-model-based-1","title":"An Analysis of Model-Based Reinforcement Learning From Abstracted Observations","date":"2022-08-30","arxiv_id":"2208.14407","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-with-sindy","title":"Model-Based Reinforcement Learning with SINDy","date":"2022-08-30","arxiv_id":"2208.14501","repositories_listed":0,"syntology":null},{"url":null,"slug":"backward-imitation-and-forward-reinforcement","title":"Backward Imitation and Forward Reinforcement Learning via Bi-directional Model Rollouts","date":"2022-08-04","arxiv_id":"2208.02434","repositories_listed":0,"syntology":null},{"url":null,"slug":"raising-student-completion-rates-with","title":"Raising Student Completion Rates with Adaptive Curriculum and Contextual Bandits","date":"2022-07-28","arxiv_id":"2207.14003","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-neural-ordinary-differential-equations","title":"Adaptive Asynchronous Control Using Meta-learned Neural Ordinary Differential Equations","date":"2022-07-25","arxiv_id":"2207.12062","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-based-model-based-reinforcement","title":"Skill-based Model-based Reinforcement Learning","date":"2022-07-15","arxiv_id":"2207.07560","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-selectively-learn-for-weakly-1","title":"Learning to Selectively Learn for Weakly Supervised Paraphrase Generation with Model-based Reinforcement Learning","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-for-1","title":"Provably Efficient Reinforcement Learning for Online Adaptive Influence Maximization","date":"2022-06-29","arxiv_id":"2206.14846","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-world-models-for-visual-control","title":"Masked World Models for Visual Control","date":"2022-06-28","arxiv_id":"2206.14244","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-voice-instructions-in-model","title":"Incorporating Voice Instructions in Model-Based Reinforcement Learning for Self-Driving Cars","date":"2022-06-21","arxiv_id":"2206.10249","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-imitation-learning-using-entropy","title":"Model-Based Imitation Learning Using Entropy Regularization of Model and Policy","date":"2022-06-21","arxiv_id":"2206.10101","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-cognitive-psychology-to-understand-gpt","title":"Using cognitive psychology to understand GPT-3","date":"2022-06-21","arxiv_id":"2206.14576","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-safe-shooting-model-based","title":"Guided Safe Shooting: model based reinforcement learning with safety constraints","date":"2022-06-20","arxiv_id":"2206.09743","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-model-based-reinforcement","title":"A Survey on Model-based Reinforcement Learning","date":"2022-06-19","arxiv_id":"2206.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-decision-time-vs-background","title":"A Look at Value-Based Decision-Time vs. Background Planning Methods Across Different Settings","date":"2022-06-16","arxiv_id":"2206.08442","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-is-minimax","title":"Model-Based Reinforcement Learning for Offline Zero-Sum Markov Games","date":"2022-06-08","arxiv_id":"2206.04044","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-model-based-reinforcement-learning-approach-1","title":"A Model-Based Reinforcement Learning Approach for PID Design","date":"2022-06-07","arxiv_id":"2206.03567","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-rollout-length-for-model-based-rl","title":"Adaptive Rollout Length for Model-Based RL Using Model-Free Deep RL","date":"2022-06-06","arxiv_id":"2206.02380","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-space-planning-with-subgoal-models","title":"Goal-Space Planning with Subgoal Models","date":"2022-06-06","arxiv_id":"2206.02902","repositories_listed":0,"syntology":null},{"url":null,"slug":"between-rate-distortion-theory-value","title":"Between Rate-Distortion Theory & Value Equivalence in Model-Based Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.02025","repositories_listed":0,"syntology":null},{"url":null,"slug":"deciding-what-to-model-value-equivalent","title":"Deciding What to Model: Value-Equivalent Sampling for Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.02072","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-causal","title":"Offline Reinforcement Learning with Causal Structured World Models","date":"2022-06-03","arxiv_id":"2206.01474","repositories_listed":0,"syntology":null},{"url":null,"slug":"posterior-coreset-construction-with","title":"Posterior Coreset Construction with Kernelized Stein Discrepancy for Model-Based Reinforcement Learning","date":"2022-06-02","arxiv_id":"2206.01162","repositories_listed":0,"syntology":null},{"url":null,"slug":"stock-trading-optimization-through-model","title":"Stock Trading Optimization through Model-based Reinforcement Learning with Resistance Support Relative Strength","date":"2022-05-30","arxiv_id":"2205.15056","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-source-transfer-learning-for-deep-model","title":"Multi-Source Transfer Learning for Deep Model-Based Reinforcement Learning","date":"2022-05-28","arxiv_id":"2205.14410","repositories_listed":0,"syntology":null},{"url":null,"slug":"should-models-be-accurate","title":"Should Models Be Accurate?","date":"2022-05-22","arxiv_id":"2205.10736","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-reinforcement-learning-for-1","title":"Accelerated Reinforcement Learning for Temporal Logic Control Objectives","date":"2022-05-09","arxiv_id":"2205.04424","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-prioritization-through-1","title":"INFOrmation Prioritization through EmPOWERment in Visual Model-Based RL","date":"2022-04-18","arxiv_id":"2204.08585","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-no-regret-model-based-meta-rl-for","title":"Online No-regret Model-Based Meta RL for Personalized Navigation","date":"2022-04-05","arxiv_id":"2204.01925","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-controller-for-output-feedback-linear","title":"Safe Controller for Output Feedback Linear Systems using Model-Based Reinforcement Learning","date":"2022-04-04","arxiv_id":"2204.01409","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-model-based-value-expansion","title":"Revisiting Model-based Value Expansion","date":"2022-03-28","arxiv_id":"2203.14660","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-compounding-prediction-errors","title":"Investigating Compounding Prediction Errors in Learned Dynamics Models","date":"2022-03-17","arxiv_id":"2203.09637","repositories_listed":0,"syntology":null},{"url":null,"slug":"dreamingv2-reinforcement-learning-with","title":"DreamingV2: Reinforcement Learning with Discrete World Models without Reconstruction","date":"2022-03-01","arxiv_id":"2203.00494","repositories_listed":0,"syntology":null},{"url":"/paper/transdreamer-reinforcement-learning-with-1","slug":"transdreamer-reinforcement-learning-with-1","title":"TransDreamer: Reinforcement Learning with Transformer World Models","date":"2022-02-19","arxiv_id":"2202.09481","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transdreamer-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2202.09481","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.09481"}},"official":null}},{"url":null,"slug":"should-i-send-this-notification-optimizing","title":"Should I send this notification? Optimizing push notifications decision making by modeling the future","date":"2022-02-17","arxiv_id":"2202.08812","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-causal-model-based","title":"Provably Efficient Causal Model-Based Reinforcement Learning for Systematic Generalization","date":"2022-02-14","arxiv_id":"2202.06545","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-nid-rules","title":"Neural NID Rules","date":"2022-02-12","arxiv_id":"2202.06036","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-respecting-subtasks-for-model-based","title":"Reward-Respecting Subtasks for Model-Based Reinforcement Learning","date":"2022-02-07","arxiv_id":"2202.03466","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-hypothesis-spaces-for","title":"Meta-Learning Hypothesis Spaces for Sequential Decision-making","date":"2022-02-01","arxiv_id":"2202.00602","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-differentiable-optimization-and","title":"Joint Differentiable Optimization and Verification for Certified Reinforcement Learning","date":"2022-01-28","arxiv_id":"2201.12243","repositories_listed":0,"syntology":null},{"url":null,"slug":"physical-derivatives-computing-policy","title":"Physical Derivatives: Computing policy gradients by physical forward-propagation","date":"2022-01-15","arxiv_id":"2201.05830","repositories_listed":0,"syntology":null},{"url":null,"slug":"opportunities-of-hybrid-model-based","title":"Opportunities of Hybrid Model-based Reinforcement Learning for Cell Therapy Manufacturing Process Control","date":"2022-01-10","arxiv_id":"2201.03116","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-policy-loss-and-planning","title":"Assessing Policy, Loss and Planning Combinations in Reinforcement Learning using a New Modular Architecture","date":"2022-01-08","arxiv_id":"2201.02874","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-exploitation-trade-off-for","title":"Exploration-exploitation trade-off for continuous-time episodic reinforcement learning with linear-convex models","date":"2021-12-19","arxiv_id":"2112.10264","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-value-inconsistency-as-a-signal-for","title":"Model-Value Inconsistency as a Signal for Epistemic Uncertainty","date":"2021-12-08","arxiv_id":"2112.04153","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-model-based-reinforcement","title":"Maximum Entropy Model-based Reinforcement Learning","date":"2021-12-02","arxiv_id":"2112.01195","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-gearshift-controllers-for-electric","title":"Improving gearshift controllers for electric vehicles with reinforcement learning","date":"2021-12-01","arxiv_id":"2112.00529","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-via","title":"Model-Based Reinforcement Learning via Imagination with Derived Memory","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-end-to-end-model-based","title":"Understanding End-to-End Model-Based Reinforcement Learning Methods as Implicit Parameterization","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"weighted-model-estimation-for-offline-model","title":"Weighted model estimation for offline model-based reinforcement learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-performance-bound-for-model-based-online","title":"Approximate infinite-horizon predictive control","date":"2021-11-16","arxiv_id":"2111.08319","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-will-belief-as-a-consequence-of-model","title":"Free Will Belief as a consequence of Model-based Reinforcement Learning","date":"2021-11-14","arxiv_id":"2111.08435","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-5","title":"Model-Based Reinforcement Learning via Stochastic Hybrid Models","date":"2021-11-11","arxiv_id":"2111.06211","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-before-you-leap-safe-model-based","title":"Look Before You Leap: Safe Model-Based Reinforcement Learning with Human Intervention","date":"2021-11-10","arxiv_id":"2111.05819","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-model-based-reinforcement","title":"Risk Sensitive Model-Based Reinforcement Learning using Uncertainty Guided Planning","date":"2021-11-09","arxiv_id":"2111.04972","repositories_listed":0,"syntology":null},{"url":"/paper/procedural-generalization-by-planning-with-1","slug":"procedural-generalization-by-planning-with-1","title":"Procedural Generalization by Planning with Self-Supervised World Models","date":"2021-11-02","arxiv_id":"2111.01587","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-multi-agent-reinforcement-1","title":"Model based Multi-agent Reinforcement Learning with Tensor Decompositions","date":"2021-10-27","arxiv_id":"2110.14524","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robust-controllers-via-probabilistic","title":"Learning Robust Controllers Via Probabilistic Model-Based Policy Search","date":"2021-10-26","arxiv_id":"2110.13576","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitask-adaptation-by-retrospective","title":"Multitask Adaptation by Retrospective Exploration with Learned World Models","date":"2021-10-25","arxiv_id":"2110.13241","repositories_listed":0,"syntology":null},{"url":null,"slug":"operator-augmentation-for-model-based-policy","title":"Operator Shifting for Model-based Policy Evaluation","date":"2021-10-25","arxiv_id":"2110.12658","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-4","title":"Model-based Reinforcement Learning for Service Mesh Fault Resiliency in a Web Application-level","date":"2021-10-21","arxiv_id":"2110.13621","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-predictive-routing-with-nested-1","title":"Variational Predictive Routing with Nested Subjective Timescales","date":"2021-10-21","arxiv_id":"2110.11236","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-hyperparameter-optimization-by-1","title":"Improving Hyperparameter Optimization by Planning Ahead","date":"2021-10-15","arxiv_id":"2110.08028","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-free-model-based-reinforcement","title":"Reward-Free Model-Based Reinforcement Learning with Linear Function Approximation","date":"2021-10-12","arxiv_id":"2110.06394","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-design-choices-in-model-based","title":"Revisiting Design Choices in Offline Model-Based Reinforcement Learning","date":"2021-10-08","arxiv_id":"2110.04135","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-reward-free-exploration-for","title":"Near-Optimal Reward-Free Exploration for Linear Mixture MDPs with Plug-in Solver","date":"2021-10-07","arxiv_id":"2110.03244","repositories_listed":0,"syntology":null},{"url":null,"slug":"imaginary-hindsight-experience-replay-curious","title":"Imaginary Hindsight Experience Replay: Curious Model-based Learning for Sparse Reward Tasks","date":"2021-10-05","arxiv_id":"2110.02414","repositories_listed":0,"syntology":null},{"url":null,"slug":"cycle-consistent-world-models-for-domain","title":"Cycle-Consistent World Models for Domain Independent Latent Imagination","date":"2021-10-02","arxiv_id":"2110.00808","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-model-based-reinforcement","title":"Safety aware model-based reinforcement learning for optimal control of a class of output-feedback nonlinear systems","date":"2021-10-01","arxiv_id":"2110.00271","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-cross-entropy-method-for-model","title":"Decentralized Cross-Entropy Method for Model-Based Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-decentralized-model-based-policy","title":"Fully Decentralized Model-based Policy Optimization with Networked Agents","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dynamics-models-for-model-predictive","title":"Learning Dynamics Models for Model Predictive Agents","date":"2021-09-29","arxiv_id":"2109.14311","repositories_listed":0,"syntology":null},{"url":null,"slug":"moba-multi-teacher-model-based-reinforcement","title":"MOBA: Multi-teacher Model Based Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-with-1","title":"Model-based Reinforcement Learning with Ensembled Model-value Expansion","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"example-driven-model-based-reinforcement","title":"Example-Driven Model-Based Reinforcement Learning for Solving Long-Horizon Visuomotor Tasks","date":"2021-09-21","arxiv_id":"2109.10312","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-emptive-learning-to-defer-for-sequential","title":"Learning-to-defer for sequential medical decision-making under uncertainty","date":"2021-09-13","arxiv_id":"2109.06312","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-ensemble-model-based-reinforcement","title":"Federated Ensemble Model-based Reinforcement Learning in Edge Computing","date":"2021-09-12","arxiv_id":"2109.05549","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-model-based-reinforcement-learning-for","title":"Robust Model-based Reinforcement Learning for Autonomous Greenhouse Control","date":"2021-08-26","arxiv_id":"2108.11645","repositories_listed":0,"syntology":null},{"url":null,"slug":"plug-and-play-model-based-reinforcement","title":"Plug and Play, Model-Based Reinforcement Learning","date":"2021-08-20","arxiv_id":"2108.08960","repositories_listed":0,"syntology":null},{"url":null,"slug":"fractional-transfer-learning-for-deep-model","title":"Fractional Transfer Learning for Deep Model-Based Reinforcement Learning","date":"2021-08-14","arxiv_id":"2108.06526","repositories_listed":0,"syntology":null},{"url":null,"slug":"mbdp-a-model-based-approach-to-achieve-both","title":"MBDP: A Model-based Approach to Achieve both Robustness and Sample Efficiency via Double Dropout Planning","date":"2021-08-03","arxiv_id":"2108.01295","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-dyna-style-model-based-deep","title":"Physics-informed Dyna-Style Model-Based Deep Reinforcement Learning for Dynamic Control","date":"2021-07-31","arxiv_id":"2108.00128","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-accuracy-model-based-reinforcement","title":"High-Accuracy Model-Based Reinforcement Learning, a Survey","date":"2021-07-17","arxiv_id":"2107.08241","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-based-multi-agent-mean-field","title":"Efficient Model-Based Multi-Agent Mean-Field Reinforcement Learning","date":"2021-07-08","arxiv_id":"2107.04050","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-control-using-learned-state-space","title":"Predictive Control Using Learned State Space Models via Rolling Horizon Evolution","date":"2021-06-25","arxiv_id":"2106.13911","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-model-based-reinforcement","title":"Uncertainty-Aware Model-Based Reinforcement Learning with Application to Autonomous Driving","date":"2021-06-23","arxiv_id":"2106.12194","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifted-model-checking-for-relational-mdps","title":"Lifted Model Checking for Relational MDPs","date":"2021-06-22","arxiv_id":"2106.11735","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperspace-neighbor-penetration-approach-to","title":"Hyperspace Neighbor Penetration Approach to Dynamic Programming for Model-Based Reinforcement Learning Problems with Slowly Changing Variables in A Continuous State Space","date":"2021-06-10","arxiv_id":"2106.05497","repositories_listed":0,"syntology":null}],"record_sha256":"17001ad175e6342ff639551eb9552fe842945c6f3cb622ef663af7d638d968e0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}