{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/offline-rl/papers/4","list_of":"/task/offline-rl","task":"Offline RL","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":8,"rows_per_page":100,"rows":[301,400],"of":755,"counts":{"archive_papers_tagged":755,"with_a_code_link":310,"where_syntology_ran_a_sample":164,"not_listed_spam_title":0,"listed":755,"listed_where_code_ran":164,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":139,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":139,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/offline-rl","prev":"/task/offline-rl/papers/3","next":"/task/offline-rl/papers/5","papers":[{"url":"/paper/q-value-weighted-regression-reinforcement-1","slug":"q-value-weighted-regression-reinforcement-1","title":"Q-Value Weighted Regression: Reinforcement Learning with Limited Data","date":"2021-02-12","arxiv_id":"2102.06782","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/q-value-weighted-regression-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2102.06782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06782"}},"official":null}},{"url":"/paper/popo-pessimistic-offline-policy-optimization","slug":"popo-pessimistic-offline-policy-optimization","title":"POPO: Pessimistic Offline Policy Optimization","date":"2020-12-26","arxiv_id":"2012.13682","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-from-images","slug":"offline-reinforcement-learning-from-images","title":"Offline Reinforcement Learning from Images with Latent Space Models","date":"2020-12-21","arxiv_id":"2012.11547","repositories_listed":1,"syntology":null},{"url":"/paper/rl-unplugged-a-collection-of-benchmarks-for","slug":"rl-unplugged-a-collection-of-benchmarks-for","title":"RL Unplugged: A Collection of Benchmarks for Offline Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/batch-exploration-with-examples-for-scalable","slug":"batch-exploration-with-examples-for-scalable","title":"Batch Exploration with Examples for Scalable Robotic Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11917","repositories_listed":1,"syntology":null},{"url":"/paper/human-centric-dialog-training-via-offline","slug":"human-centric-dialog-training-via-offline","title":"Human-centric Dialog Training via Offline Reinforcement Learning","date":"2020-10-12","arxiv_id":"2010.05848","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-dialog-training-via-offline#ran","syntology_url":"https://syntology.ai/paper/2010.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05848"}},"official":{"repos":["natashamjaques/neural_chat"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/efficient-fully-offline-meta-reinforcement","slug":"efficient-fully-offline-meta-reinforcement","title":"FOCAL: Efficient Fully-Offline Meta-Reinforcement Learning via Distance Metric Learning and Behavior Regularization","date":"2020-10-02","arxiv_id":"2010.01112","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-fully-offline-meta-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2010.01112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.01112"}},"official":{"repos":["FOCAL-ICLR/FOCAL-ICLR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-optimistic-perspective-on-offline-deep","slug":"an-optimistic-perspective-on-offline-deep","title":"An Optimistic Perspective on Offline Deep Reinforcement Learning","date":"2020-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/behavior-regularized-offline-reinforcement-1","slug":"behavior-regularized-offline-reinforcement-1","title":"Behavior Regularized Offline Reinforcement Learning","date":"2019-11-26","arxiv_id":"1911.11361","repositories_listed":1,"syntology":null},{"url":"/paper/striving-for-simplicity-in-off-policy-deep","slug":"striving-for-simplicity-in-off-policy-deep","title":"An Optimistic Perspective on Offline Reinforcement Learning","date":"2019-07-10","arxiv_id":"1907.04543","repositories_listed":1,"syntology":null},{"url":null,"slug":"from-novelty-to-imitation-self-distilled","title":"From Novelty to Imitation: Self-Distilled Rewards for Offline Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.12815","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-bandwidth-estimation-for-real-time","title":"Robust Bandwidth Estimation for Real-Time Communication with Offline Reinforcement Learning","date":"2025-07-08","arxiv_id":"2507.05785","repositories_listed":0,"syntology":null},{"url":"/paper/flow-based-single-step-completion-for","slug":"flow-based-single-step-completion-for","title":"Flow-Based Single-Step Completion for Efficient and Expressive Policy Learning","date":"2025-06-26","arxiv_id":"2506.21427","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flow-based-single-step-completion-for#ran","syntology_url":"https://syntology.ai/paper/2506.21427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21427"}},"official":null}},{"url":null,"slug":"optimal-single-policy-sample-complexity-and","title":"Optimal Single-Policy Sample Complexity and Transient Coverage for Average-Reward Offline RL","date":"2025-06-26","arxiv_id":"2506.20904","repositories_listed":0,"syntology":null},{"url":null,"slug":"intellilung-advancing-safe-mechanical","title":"IntelliLung: Advancing Safe Mechanical Ventilation using Offline RL with Hybrid Actions and Clinically Aligned Rewards","date":"2025-06-17","arxiv_id":"2506.14375","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-explainable-offline-rl-analyzing","title":"Toward Explainable Offline RL: Analyzing Representations in Intrinsically Motivated Decision Transformers","date":"2025-06-16","arxiv_id":"2506.13958","repositories_listed":0,"syntology":null},{"url":null,"slug":"moorl-a-framework-for-integrating-offline","title":"MOORL: A Framework for Integrating Offline-Online Reinforcement Learning","date":"2025-06-11","arxiv_id":"2506.09574","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08463","title":"How to Provably Improve Return Conditioned Supervised Learning?","date":"2025-06-10","arxiv_id":"2506.08463","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-based-trajectory-clustering-in-offline","title":"Policy-Based Trajectory Clustering in Offline Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.09202","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-gradient-dice-for-offline-constrained","title":"Semi-gradient DICE for Offline Constrained Reinforcement Learning","date":"2025-06-10","arxiv_id":"2506.08644","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-diffusion-models-in-offline-rl","title":"Accelerating Diffusion Models in Offline RL via Reward-Aware Consistency Trajectory Distillation","date":"2025-06-09","arxiv_id":"2506.07822","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-clarify-by-reinforcement-learning","title":"Learning to Clarify by Reinforcement Learning Through Reward-Weighted Fine-Tuning","date":"2025-06-08","arxiv_id":"2506.06964","repositories_listed":0,"syntology":null},{"url":null,"slug":"adg-ambient-diffusion-guided-dataset-recovery","title":"ADG: Ambient Diffusion-Guided Dataset Recovery for Corruption-Robust Offline Reinforcement Learning","date":"2025-05-29","arxiv_id":"2505.23871","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-dacer-algorithm-with-high-diffusion","title":"Enhanced DACER Algorithm with High Diffusion Efficiency","date":"2025-05-29","arxiv_id":"2505.23426","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-offline-rl-via-efficient-and","title":"Scaling Offline RL via Efficient and Expressive Shortcut Models","date":"2025-05-28","arxiv_id":"2505.22866","repositories_listed":0,"syntology":null},{"url":null,"slug":"genpo-generative-diffusion-models-meet-on","title":"GenPO: Generative Diffusion Models Meet On-Policy Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18763","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-self-weighted-guidance-for-offline","title":"Diffusion Self-Weighted Guidance for Offline Reinforcement Learning","date":"2025-05-23","arxiv_id":"2505.18345","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-online-rl-fine-tuning-with-offline","title":"Efficient Online RL Fine Tuning with Offline Pre-trained Policy Only","date":"2025-05-22","arxiv_id":"2505.16856","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-guarded-safe-reinforcement-learning","title":"Offline Guarded Safe Reinforcement Learning for Medical Treatment Optimization Strategies","date":"2025-05-22","arxiv_id":"2505.16242","repositories_listed":0,"syntology":null},{"url":null,"slug":"unearthing-gems-from-stones-policy","title":"Unearthing Gems from Stones: Policy Optimization with Negative Sample Augmentation for LLM Reasoning","date":"2025-05-20","arxiv_id":"2505.14403","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-offline-policy-is-not-trustworthy","title":"Your Offline Policy is Not Trustworthy: Bilevel Reinforcement Learning for Sequential Portfolio Optimization","date":"2025-05-19","arxiv_id":"2505.12759","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10881","title":"Prior-Guided Diffusion Planning for Offline Reinforcement Learning","date":"2025-05-16","arxiv_id":"2505.10881","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-individual-optimal","title":"Reinforcement Learning for Individual Optimal Policy from Heterogeneous Data","date":"2025-05-14","arxiv_id":"2505.09496","repositories_listed":0,"syntology":null},{"url":null,"slug":"feasibility-aware-pessimistic-estimation","title":"Feasibility-Aware Pessimistic Estimation: Toward Long-Horizon Safety in Offline RL","date":"2025-05-13","arxiv_id":"2505.08179","repositories_listed":0,"syntology":null},{"url":null,"slug":"cache-efficient-posterior-sampling-for","title":"Cache-Efficient Posterior Sampling for Reinforcement Learning with LLM-Derived Priors Across Discrete and Continuous Domains","date":"2025-05-12","arxiv_id":"2505.07274","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-matters-for-batch-online-reinforcement","title":"What Matters for Batch Online Reinforcement Learning in Robotics?","date":"2025-05-12","arxiv_id":"2505.08078","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-enhanced-offline-reinforcement-learning","title":"Video-Enhanced Offline Reinforcement Learning: A Model-Based Approach","date":"2025-05-10","arxiv_id":"2505.06482","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretraining-a-shared-q-network-for-data","title":"Pretraining a Shared Q-Network for Data-Efficient Offline Reinforcement Learning","date":"2025-05-09","arxiv_id":"2505.05701","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-ood-actions-for-offline-reinforcement","title":"Taming OOD Actions for Offline Reinforcement Learning: An Advantage-Based Approach","date":"2025-05-08","arxiv_id":"2505.05126","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-potential-of-offline-rl-for","title":"Exploring the Potential of Offline RL for Reasoning in LLMs: A Preliminary Study","date":"2025-05-04","arxiv_id":"2505.02142","repositories_listed":0,"syntology":null},{"url":null,"slug":"analytic-energy-guided-policy-optimization","title":"Analytic Energy-Guided Policy Optimization for Offline Reinforcement Learning","date":"2025-05-03","arxiv_id":"2505.01822","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-robotic-world-model-learning-robotic","title":"Offline Robotic World Model: Learning Robotic Policies without a Physics Simulator","date":"2025-04-23","arxiv_id":"2504.16680","repositories_listed":0,"syntology":null},{"url":"/paper/vipo-value-function-inconsistency-penalized","slug":"vipo-value-function-inconsistency-penalized","title":"VIPO: Value Function Inconsistency Penalized Offline Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.11944","repositories_listed":0,"syntology":{"n":14,"n_ran":10,"n_constructed":9,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":14,"phrase":"10 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vipo-value-function-inconsistency-penalized#ran","syntology_url":"https://syntology.ai/paper/2504.11944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11944"}},"official":null}},{"url":null,"slug":"towards-optimal-differentially-private-regret","title":"Towards Optimal Differentially Private Regret Bounds in Linear MDPs","date":"2025-04-12","arxiv_id":"2504.09339","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-spikeformer-spike-driven-transformer","title":"Decision SpikeFormer: Spike-Driven Transformer for Decision Making","date":"2025-04-04","arxiv_id":"2504.03800","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-reinforcement-learning-3","title":"Model-Based Offline Reinforcement Learning with Adversarial Data Augmentation","date":"2025-03-26","arxiv_id":"2503.20285","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-discrete","title":"Offline Reinforcement Learning with Discrete Diffusion Skills","date":"2025-03-26","arxiv_id":"2503.20176","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviour-discovery-and-attribution-for","title":"Behaviour Discovery and Attribution for Explainable Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.14973","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-time-policy-switching-for-offline","title":"Evaluation-Time Policy Switching for Offline Reinforcement Learning","date":"2025-03-15","arxiv_id":"2503.12222","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-pitfalls-of-imitation-learning-when","title":"The Pitfalls of Imitation Learning when Actions are Continuous","date":"2025-03-12","arxiv_id":"2503.09722","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-regularization-on-globally-accessible","title":"Policy Regularization on Globally Accessible States in Cross-Dynamics Reinforcement Learning","date":"2025-03-10","arxiv_id":"2503.06893","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-weighted-flow-matching-for-offline","title":"Energy-Weighted Flow Matching for Offline Reinforcement Learning","date":"2025-03-06","arxiv_id":"2503.04975","repositories_listed":0,"syntology":null},{"url":null,"slug":"yes-q-learning-helps-offline-in-context-rl","title":"Yes, Q-learning Helps Offline In-Context RL","date":"2025-02-24","arxiv_id":"2502.17666","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-offline-model-based-rl-via-active","title":"Enhancing Offline Model-Based RL via Active Model Selection: A Bayesian Optimization Perspective","date":"2025-02-17","arxiv_id":"2502.11480","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-features-are-best-for-successor","title":"Which Features are Best for Successor Features?","date":"2025-02-15","arxiv_id":"2502.10790","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-transformer-decoding-for-offline","title":"Diverse Transformer Decoding for Offline Reinforcement Learning Using Financial Algorithmic Approaches","date":"2025-02-13","arxiv_id":"2502.10473","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-pre-trained-decision-transformers","title":"Enhancing Pre-Trained Decision Transformers with Prompt-Tuning Bandits","date":"2025-02-07","arxiv_id":"2502.04979","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-entropy-guided-dataset-generation","title":"Behavioral Entropy-Guided Dataset Generation for Offline Reinforcement Learning","date":"2025-02-06","arxiv_id":"2502.04141","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnirl-in-context-reinforcement-learning-by","title":"OmniRL: In-Context Reinforcement Learning by Large-Scale Meta-Training in Randomized Worlds","date":"2025-02-05","arxiv_id":"2502.02869","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-guided-causal-state-representation-for","title":"Policy-Guided Causal State Representation for Offline Reinforcement Learning Recommendation","date":"2025-02-04","arxiv_id":"2502.02327","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-uav-trajectory-planning-via-few","title":"Resilient UAV Trajectory Planning via Few-Shot Meta-Offline Reinforcement Learning","date":"2025-02-03","arxiv_id":"2502.01268","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-blood-glucose-control-offline","title":"Flexible Blood Glucose Control: Offline Reinforcement Learning from Human Feedback","date":"2025-01-27","arxiv_id":"2501.15972","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-center-cooling-system-optimization-using","title":"Data Center Cooling System Optimization Using Offline Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15085","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-driven-policy","title":"Large Language Model driven Policy Exploration for Recommender Systems","date":"2025-01-23","arxiv_id":"2501.13816","repositories_listed":0,"syntology":null},{"url":null,"slug":"drdt3-diffusion-refined-decision-test-time","title":"DRDT3: Diffusion-Refined Decision Test-Time Training Model","date":"2025-01-12","arxiv_id":"2501.06718","repositories_listed":0,"syntology":null},{"url":null,"slug":"sr-reward-taking-the-path-more-traveled","title":"SR-Reward: Taking The Path More Traveled","date":"2025-01-04","arxiv_id":"2501.02330","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-statistical-complexity-for-offline-and","title":"On the Statistical Complexity for Offline and Low-Adaptive Reinforcement Learning with Structures","date":"2025-01-03","arxiv_id":"2501.02089","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-data-augmentation-for","title":"Goal-Conditioned Data Augmentation for Offline Reinforcement Learning","date":"2024-12-29","arxiv_id":"2412.20519","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multi-step-reasoning-abilities-of","title":"Improving Multi-Step Reasoning Abilities of Large Language Models with Direct Advantage Policy Optimization","date":"2024-12-24","arxiv_id":"2412.18279","repositories_listed":0,"syntology":null},{"url":null,"slug":"adacred-adaptive-causal-decision-transformers","title":"AdaCred: Adaptive Causal Decision Transformers with Feature Crediting","date":"2024-12-19","arxiv_id":"2412.15427","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-agnostic-rl-offline-rl-and-online-rl","title":"Policy Agnostic RL: Offline RL and Online RL Fine-Tuning of Any Class and Backbone","date":"2024-12-09","arxiv_id":"2412.06685","repositories_listed":0,"syntology":null},{"url":null,"slug":"finer-behavioral-foundation-models-via-auto","title":"Finer Behavioral Foundation Models via Auto-Regressive Features and Advantage Weighting","date":"2024-12-05","arxiv_id":"2412.04368","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-dynamic-object-interactions-in-text","title":"Improving Dynamic Object Interactions in Text-to-Video Generation with AI Feedback","date":"2024-12-03","arxiv_id":"2412.02617","repositories_listed":0,"syntology":null},{"url":"/paper/robust-offline-reinforcement-learning-with-1","slug":"robust-offline-reinforcement-learning-with-1","title":"Robust Offline Reinforcement Learning with Linearly Structured $f$-Divergence Regularization","date":"2024-11-27","arxiv_id":"2411.18612","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-offline-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2411.18612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18612"}},"official":null}},{"url":null,"slug":"llm-based-offline-learning-for-embodied","title":"LLM-Based Offline Learning for Embodied Agents via Consistency-Guided Reward Ensemble","date":"2024-11-26","arxiv_id":"2411.17135","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressor-a-perceptually-guided-reward","title":"PROGRESSOR: A Perceptually Guided Reward Estimator with Self-Supervised Online Refinement","date":"2024-11-26","arxiv_id":"2411.17764","repositories_listed":0,"syntology":null},{"url":null,"slug":"preserving-expert-level-privacy-in-offline","title":"Preserving Expert-Level Privacy in Offline Reinforcement Learning","date":"2024-11-18","arxiv_id":"2411.13598","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-with-qphil-quantizing-planner-for","title":"Navigation with QPHIL: Quantizing Planner for Hierarchical Implicit Q-Learning","date":"2024-11-12","arxiv_id":"2411.07760","repositories_listed":0,"syntology":null},{"url":null,"slug":"streetwise-agents-empowering-offline-rl","title":"Streetwise Agents: Empowering Offline RL Policies to Outsmart Exogenous Stochastic Disturbances in RTC","date":"2024-11-11","arxiv_id":"2411.06815","repositories_listed":0,"syntology":null},{"url":null,"slug":"offlight-an-offline-multi-agent-reinforcement","title":"OffLight: An Offline Multi-Agent Reinforcement Learning Framework for Traffic Signal Control","date":"2024-11-10","arxiv_id":"2411.06601","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-offline-reinforcement-learning-1","title":"Real-World Offline Reinforcement Learning from Vision Language Model Feedback","date":"2024-11-08","arxiv_id":"2411.05273","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-sft-q-learning-for-language-models-via","title":"Q-SFT: Q-Learning for Language Models via Supervised Fine-Tuning","date":"2024-11-07","arxiv_id":"2411.05193","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-and-sequence","title":"Offline Reinforcement Learning and Sequence Modeling for Downlink Link Adaptation","date":"2024-10-30","arxiv_id":"2410.23031","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-job-shop","title":"Offline reinforcement learning for job-shop scheduling problems","date":"2024-10-21","arxiv_id":"2410.15714","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-continual-offline-rl-through","title":"Solving Continual Offline RL through Selective Weights Activation on Aligned Spaces","date":"2024-10-21","arxiv_id":"2410.15698","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-dynamics-conditional-diffusion-planners","title":"Off-dynamics Conditional Diffusion Planners","date":"2024-10-16","arxiv_id":"2410.12238","repositories_listed":0,"syntology":null},{"url":null,"slug":"diar-diffusion-model-guided-implicit-q","title":"DIAR: Diffusion-model-guided Implicit Q-learning with Adaptive Revaluation","date":"2024-10-15","arxiv_id":"2410.11338","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-offline-rl-for-improved","title":"Diffusion-Based Offline RL for Improved Decision-Making in Augmented ARC Task","date":"2024-10-15","arxiv_id":"2410.11324","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-optimization-multi-auv","title":"Multi-Objective-Optimization Multi-AUV Assisted Data Collection Framework for IoUT Based on Offline Reinforcement Learning","date":"2024-10-15","arxiv_id":"2410.11282","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-reinforcement-learning-and-large","title":"Integrating Reinforcement Learning and Large Language Models for Crop Production Process Management Optimization and Control through A New Knowledge-Based Deep Learning Paradigm","date":"2024-10-13","arxiv_id":"2410.09680","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-inverse-constrained-reinforcement","title":"Offline Inverse Constrained Reinforcement Learning for Safe-Critical Decision Making in Healthcare","date":"2024-10-10","arxiv_id":"2410.07525","repositories_listed":0,"syntology":null},{"url":null,"slug":"comadice-offline-cooperative-multi-agent","title":"ComaDICE: Offline Cooperative Multi-Agent Reinforcement Learning with Stationary Distribution Shift Regularization","date":"2024-10-02","arxiv_id":"2410.01954","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-data-and-calibrated-simulation","title":"The Smart Buildings Control Suite: A Diverse Open Source Benchmark to Evaluate and Scale HVAC Control Policies for Sustainability","date":"2024-10-02","arxiv_id":"2410.03756","repositories_listed":0,"syntology":null},{"url":null,"slug":"offripp-offline-rl-based-informative-path","title":"OffRIPP: Offline RL-based Informative Path Planning","date":"2024-09-25","arxiv_id":"2409.16830","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-and-validation-of-heparin-dosing","title":"Development and Validation of Heparin Dosing Policies Using an Offline Reinforcement Learning Algorithm","date":"2024-09-24","arxiv_id":"2409.15753","repositories_listed":0,"syntology":null},{"url":null,"slug":"kan-v-s-mlp-for-offline-reinforcement","title":"KAN v.s. MLP for Offline Reinforcement Learning","date":"2024-09-15","arxiv_id":"2409.09653","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-value-regularized-decision-convformer-for","title":"Q-value Regularized Decision ConvFormer for Offline Reinforcement Learning","date":"2024-09-12","arxiv_id":"2409.08062","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-cross-domain-pre-trained-decision","title":"Enhancing Cross-domain Pre-Trained Decision Transformers with Adaptive Attention","date":"2024-09-11","arxiv_id":"2409.06985","repositories_listed":0,"syntology":null},{"url":null,"slug":"tractable-offline-learning-of-regular","title":"Tractable Offline Learning of Regular Decision Processes","date":"2024-09-04","arxiv_id":"2409.02747","repositories_listed":0,"syntology":null},{"url":null,"slug":"skills-regularized-task-decomposition-for","title":"Skills Regularized Task Decomposition for Multi-task Offline Reinforcement Learning","date":"2024-08-28","arxiv_id":"2408.15593","repositories_listed":0,"syntology":null}],"record_sha256":"c7bbc4ed1f3ded8c98feee4dfcf51b42ba372ef4226314191f82e25f00755aa6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}