{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/offline-rl/papers/6","list_of":"/task/offline-rl","task":"Offline RL","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":8,"rows_per_page":100,"rows":[501,600],"of":755,"counts":{"archive_papers_tagged":755,"with_a_code_link":310,"where_syntology_ran_a_sample":164,"not_listed_spam_title":0,"listed":755,"listed_where_code_ran":164,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":139,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":139,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/offline-rl","prev":"/task/offline-rl/papers/5","next":"/task/offline-rl/papers/7","papers":[{"url":null,"slug":"self-driving-telescopes-autonomous-scheduling","title":"Self-Driving Telescopes: Autonomous Scheduling of Astronomical Observation Campaigns with Offline Reinforcement Learning","date":"2023-11-29","arxiv_id":"2311.18094","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-fully-data-driven-approach-for-realistic","title":"A Fully Data-Driven Approach for Realistic Traffic Signal Control Using Offline Reinforcement Learning","date":"2023-11-27","arxiv_id":"2311.15920","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-wireless","title":"Offline Reinforcement Learning for Wireless Network Optimization with Mixture Datasets","date":"2023-11-19","arxiv_id":"2311.11423","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-decision-transformer-via","title":"Rethinking Decision Transformer via Hierarchical Reinforcement Learning","date":"2023-11-01","arxiv_id":"2311.00267","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-modeling-is-insufficient-for","title":"A Tractable Inference Perspective of Offline RL","date":"2023-10-31","arxiv_id":"2311.00094","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-rl-with-observation-histories","title":"Offline RL with Observation Histories: Analyzing and Improving Sample Complexity","date":"2023-10-31","arxiv_id":"2310.20663","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-causal-representation-for","title":"Safety-aware Causal Representation for Trustworthy Offline Reinforcement Learning in Autonomous Driving","date":"2023-10-31","arxiv_id":"2311.10747","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-data-augmentation-for-offline","title":"Guided Data Augmentation for Offline Reinforcement Learning and Imitation Learning","date":"2023-10-27","arxiv_id":"2310.18247","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-enhanced-contrastive-reinforcement","title":"Model-enhanced Contrastive Reinforcement Learning for Sequential Recommendation","date":"2023-10-25","arxiv_id":"2310.16566","repositories_listed":0,"syntology":null},{"url":null,"slug":"finetuning-offline-world-models-in-the-real","title":"Finetuning Offline World Models in the Real World","date":"2023-10-24","arxiv_id":"2310.16029","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-quantized-offline-reinforcement","title":"Action-Quantized Offline Reinforcement Learning for Robotic Skill Learning","date":"2023-10-18","arxiv_id":"2310.11731","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-offline-reinforcement-learning-for","title":"End-to-end Offline Reinforcement Learning for Glycemia Control","date":"2023-10-16","arxiv_id":"2310.10312","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-optimal-transport-for-enhanced","title":"Leveraging Optimal Transport for Enhanced Offline Reinforcement Learning in Surgical Robotic Environments","date":"2023-10-13","arxiv_id":"2310.08841","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-level-offline-policy-optimization-with","title":"Bi-Level Offline Policy Optimization with Limited Exploration","date":"2023-10-10","arxiv_id":"2310.06268","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-to-go-out-of-distribution-in-offline","title":"Planning to Go Out-of-Distribution in Offline-to-Online Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05723","repositories_listed":0,"syntology":null},{"url":null,"slug":"sera-sample-efficient-reward-augmentation-in","title":"Improving Offline-to-Online Reinforcement Learning with Q Conditioned State Entropy Exploration","date":"2023-10-07","arxiv_id":"2310.19805","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-confirming-transformer-for-locally","title":"Self-Confirming Transformer for Belief-Conditioned Adaptation in Offline Multi-Agent Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04579","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-nonlinear-least-squares-value","title":"Pessimistic Nonlinear Least-Squares Value Iteration for Offline Reinforcement Learning","date":"2023-10-02","arxiv_id":"2310.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-decision-transformer-for","title":"Uncertainty-Aware Decision Transformer for Stochastic Driving Environments","date":"2023-09-28","arxiv_id":"2309.16397","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-offline-reinforcement-learning-for","title":"Boosting Offline Reinforcement Learning for Autonomous Driving with Hierarchical Latent Skills","date":"2023-09-24","arxiv_id":"2309.13614","repositories_listed":0,"syntology":null},{"url":null,"slug":"h2o-an-improved-framework-for-hybrid-offline","title":"H2O+: An Improved Framework for Hybrid Offline-and-Online RL with Dynamics Gaps","date":"2023-09-22","arxiv_id":"2309.12716","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-offline-rl-from-internet-videos-via","title":"Robotic Offline RL from Internet Videos via Value-Function Pre-Training","date":"2023-09-22","arxiv_id":"2309.13041","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-transformer-scalable-offline-reinforcement","title":"Q-Transformer: Scalable Offline Reinforcement Learning via Autoregressive Q-Functions","date":"2023-09-18","arxiv_id":"2309.10150","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-mildly-conservative-model-based","title":"DOMAIN: MilDly COnservative Model-BAsed OfflINe Reinforcement Learning","date":"2023-09-16","arxiv_id":"2309.08925","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-data-augmentation-for","title":"Equivariant Data Augmentation for Generalization in Offline Reinforcement Learning","date":"2023-09-14","arxiv_id":"2309.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"hundreds-guide-millions-adaptive-offline","title":"Hundreds Guide Millions: Adaptive Offline Reinforcement Learning with Expert Guidance","date":"2023-09-04","arxiv_id":"2309.01448","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-decision-transformers-for","title":"Multi-Objective Decision Transformers for Offline Reinforcement Learning","date":"2023-08-31","arxiv_id":"2308.16379","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-self-training-rest-for-language","title":"Reinforced Self-Training (ReST) for Language Modeling","date":"2023-08-17","arxiv_id":"2308.08998","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-robot-challenge-2022-learning-dexterous","title":"Real Robot Challenge 2022: Learning Dexterous Manipulation from Offline Data in the Real World","date":"2023-08-15","arxiv_id":"2308.07741","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-generalization-in-offline","title":"Exploiting Generalization in Offline Reinforcement Learning via Unseen State Augmentations","date":"2023-08-07","arxiv_id":"2308.03882","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-offline-reinforcement-learning","title":"Integrating Offline Reinforcement Learning with Transformers for Sequential Recommendation","date":"2023-07-26","arxiv_id":"2307.14450","repositories_listed":0,"syntology":null},{"url":null,"slug":"pasta-pretrained-action-state-transformer","title":"PASTA: Pretrained Action-State Transformer Agents","date":"2023-07-20","arxiv_id":"2307.10936","repositories_listed":0,"syntology":null},{"url":null,"slug":"budgeting-counterfactual-for-offline-rl","title":"Budgeting Counterfactual for Offline RL","date":"2023-07-12","arxiv_id":"2307.06328","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-policies-for-out-of-distribution","title":"Diffusion Policies for Out-of-Distribution Generalization in Offline Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04726","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-predictive-coding-as-an","title":"Goal-Conditioned Predictive Coding for Offline Reinforcement Learning","date":"2023-07-07","arxiv_id":"2307.03406","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-7","title":"Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-07-06","arxiv_id":"2307.02752","repositories_listed":0,"syntology":null},{"url":null,"slug":"llql-logistic-likelihood-q-learning-for","title":"LLQL: Logistic Likelihood Q-Learning for Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02345","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritized-trajectory-replay-a-replay-memory","title":"Prioritized Trajectory Replay: A Replay Memory for Data-driven Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15503","repositories_listed":0,"syntology":null},{"url":null,"slug":"chipformer-transferable-chip-placement-via","title":"ChiPFormer: Transferable Chip Placement via Offline Decision Transformer","date":"2023-06-26","arxiv_id":"2306.14744","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-from-policies-conservative-test-time-1","title":"Design from Policies: Conservative Test-Time Adaptation for Offline Policy Optimization","date":"2023-06-26","arxiv_id":"2306.14479","repositories_listed":0,"syntology":null},{"url":null,"slug":"fighting-uncertainty-with-gradients-offline","title":"Fighting Uncertainty with Gradients: Offline Reinforcement Learning via Diffusion Score Matching","date":"2023-06-24","arxiv_id":"2306.14079","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-policy-evaluation-for-reinforcement","title":"Offline Policy Evaluation for Reinforcement Learning with Adaptively Collected Data","date":"2023-06-24","arxiv_id":"2306.14063","repositories_listed":0,"syntology":null},{"url":null,"slug":"clue-calibrated-latent-guidance-for-offline","title":"CLUE: Calibrated Latent Guidance for Offline Reinforcement Learning","date":"2023-06-23","arxiv_id":"2306.13412","repositories_listed":0,"syntology":null},{"url":null,"slug":"warm-start-actor-critic-from-approximation","title":"Warm-Start Actor-Critic: From Approximation Error to Sub-optimality Gap","date":"2023-06-20","arxiv_id":"2306.11271","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-trade-off-adaptation-in-offline-rl","title":"Automatic Trade-off Adaptation in Offline RL","date":"2023-06-16","arxiv_id":"2306.09744","repositories_listed":0,"syntology":null},{"url":null,"slug":"pi2-text-vec-policy-representations-with","title":"$\\pi2\\text{vec}$: Policy Representations with Successor Features","date":"2023-06-16","arxiv_id":"2306.09800","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-multi-agent-reinforcement-learning","title":"Offline Multi-Agent Reinforcement Learning with Coupled Value Factorization","date":"2023-06-15","arxiv_id":"2306.08900","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-offline-reinforcement-1","title":"Provably Efficient Offline Reinforcement Learning with Perturbed Data Sources","date":"2023-06-14","arxiv_id":"2306.08364","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-unified-uncertainty-guided-framework","title":"A Simple Unified Uncertainty-Guided Framework for Offline-to-Online Reinforcement Learning","date":"2023-06-13","arxiv_id":"2306.07541","repositories_listed":0,"syntology":null},{"url":null,"slug":"pruning-the-way-to-reliable-policies-a-multi","title":"Pruning the Way to Reliable Policies: A Multi-Objective Deep Q-Learning Approach to Critical Care","date":"2023-06-13","arxiv_id":"2306.08044","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-based-offline-to-online","title":"ENOTO: Improving Offline-to-Online Reinforcement Learning with Q-Ensembles","date":"2023-06-12","arxiv_id":"2306.06871","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-sample-policy-iteration-for-offline","title":"Iteratively Refined Behavior Regularization for Offline Reinforcement Learning","date":"2023-06-09","arxiv_id":"2306.05726","repositories_listed":0,"syntology":null},{"url":null,"slug":"instructed-diffuser-with-temporal-condition","title":"Instructed Diffuser with Temporal Condition Guidance for Offline Reinforcement Learning","date":"2023-06-08","arxiv_id":"2306.04875","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-regularized-policy-optimization-on-data","title":"State Regularized Policy Optimization on Data with Dynamics Shift","date":"2023-06-06","arxiv_id":"2306.03552","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-preference-learning-for-offline-rl","title":"PEARL: Zero-shot Cross-task Preference Alignment and Robust Reward Learning for Robotic Manipulation","date":"2023-06-06","arxiv_id":"2306.03615","repositories_listed":0,"syntology":null},{"url":null,"slug":"survival-instinct-in-offline-reinforcement","title":"Survival Instinct in Offline Reinforcement Learning","date":"2023-06-05","arxiv_id":"2306.03286","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-fairness-in-multi-agent-markov","title":"Achieving Fairness in Multi-Agent Markov Decision Processes Using Reinforcement Learning","date":"2023-06-01","arxiv_id":"2306.00324","repositories_listed":0,"syntology":null},{"url":null,"slug":"delphic-offline-reinforcement-learning-under","title":"Delphic Offline Reinforcement Learning under Nonidentifiable Hidden Confounding","date":"2023-06-01","arxiv_id":"2306.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-offline-rl-by-blending-heuristics","title":"Improving Offline RL by Blending Heuristics","date":"2023-06-01","arxiv_id":"2306.00321","repositories_listed":0,"syntology":null},{"url":null,"slug":"iql-td-mpc-implicit-q-learning-for","title":"IQL-TD-MPC: Implicit Q-Learning for Hierarchical Model Predictive Control","date":"2023-06-01","arxiv_id":"2306.00867","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-human-feedback","title":"Reinforcement Learning with Human Feedback: Learning Dynamic Choices via Pessimism","date":"2023-05-29","arxiv_id":"2305.18438","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-primal-dual-reinforcement-learning","title":"Offline Primal-Dual Reinforcement Learning for Linear MDPs","date":"2023-05-22","arxiv_id":"2305.12944","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-6","title":"Offline Reinforcement Learning with Additional Covering Distributions","date":"2023-05-22","arxiv_id":"2305.12679","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-reparameterization-of-reward","title":"Bayesian Reparameterization of Reward-Conditioned Reinforcement Learning with Energy-based Models","date":"2023-05-18","arxiv_id":"2305.11340","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-agnostic-fine-tuning-provable","title":"Reward-agnostic Fine-tuning: Provable Statistical Benefits of Hybrid Reinforcement Learning","date":"2023-05-17","arxiv_id":"2305.10282","repositories_listed":0,"syntology":null},{"url":null,"slug":"slic-hf-sequence-likelihood-calibration-with","title":"SLiC-HF: Sequence Likelihood Calibration with Human Feedback","date":"2023-05-17","arxiv_id":"2305.10425","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-pessimism-is-provably-efficient-for","title":"Double Pessimism is Provably Efficient for Distributionally Robust Offline Reinforcement Learning: Generic Algorithm and Robust Partial Coverage","date":"2023-05-16","arxiv_id":"2305.09659","repositories_listed":0,"syntology":null},{"url":"/paper/towards-generalizable-reinforcement-learning","slug":"towards-generalizable-reinforcement-learning","title":"Towards Generalizable Reinforcement Learning for Trade Execution","date":"2023-05-12","arxiv_id":"2307.11685","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/towards-generalizable-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2307.11685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11685"}},"official":null}},{"url":null,"slug":"provable-benefits-of-general-coverage","title":"What can online reinforcement learning with function approximation benefit from general coverage conditions?","date":"2023-04-25","arxiv_id":"2304.12886","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-optimal-reward-agnostic-exploration","title":"Minimax-Optimal Reward-Agnostic Exploration in Reinforcement Learning","date":"2023-04-14","arxiv_id":"2304.07278","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-a-network-ai-gym-for-autonomous","title":"Enabling A Network AI Gym for Autonomous Cyber Agents","date":"2023-04-03","arxiv_id":"2304.01366","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-emulation-simulation-training","title":"Unified Emulation-Simulation Training Environment for Autonomous Cyber Agents","date":"2023-04-03","arxiv_id":"2304.01244","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-reinforcement-learning","title":"Understanding Reinforcement Learning Algorithms: The Progress from Basic Q-learning to Proximal Policy Optimization","date":"2023-03-31","arxiv_id":"2304.00026","repositories_listed":0,"syntology":null},{"url":null,"slug":"finetuning-from-offline-reinforcement","title":"Finetuning from Offline Reinforcement Learning: Challenges, Trade-offs and Practical Solutions","date":"2023-03-30","arxiv_id":"2303.17396","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-rl-with-hierarchical-action-exploration","title":"Deep RL with Hierarchical Action Exploration for Dialogue Generation","date":"2023-03-22","arxiv_id":"2303.13465","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-policy-learning-for-offline-to","title":"Adaptive Policy Learning for Offline-to-Online Reinforcement Learning","date":"2023-03-14","arxiv_id":"2303.07693","repositories_listed":0,"syntology":null},{"url":null,"slug":"deploying-offline-reinforcement-learning-with","title":"Deploying Offline Reinforcement Learning with Human Feedback","date":"2023-03-13","arxiv_id":"2303.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-environment-transformer-and-offline","title":"Environment Transformer and Policy Optimization for Model-Based Offline Reinforcement Learning","date":"2023-03-07","arxiv_id":"2303.03811","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-decision-transformer","title":"Graph Decision Transformer","date":"2023-03-07","arxiv_id":"2303.03747","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-of-vanilla-model","title":"On the Sample Complexity of Vanilla Model-Based Offline Reinforcement Learning with Dependent Samples","date":"2023-03-07","arxiv_id":"2303.04268","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-influence-human-behavior-with","title":"Learning to Influence Human Behavior with Offline Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.02265","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-provable-benefits-of-unsupervised-data","title":"The Provable Benefits of Unsupervised Data Sharing for Offline Reinforcement Learning","date":"2023-02-27","arxiv_id":"2302.13493","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-gauss-newton-temporal","title":"Gauss-Newton Temporal Difference Learning with Nonlinear Function Approximation","date":"2023-02-25","arxiv_id":"2302.13087","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-offline-reinforcement-learning-for-real","title":"Deep Offline Reinforcement Learning for Real-world Treatment Optimization Applications","date":"2023-02-15","arxiv_id":"2302.07549","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-context-language-decision-transformers","title":"Language Decision Transformers with Exponential Tilt for Interactive Text Environments","date":"2023-02-10","arxiv_id":"2302.05507","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-strong-baseline-for-batch-imitation","title":"A Strong Baseline for Batch Imitation Learning","date":"2023-02-06","arxiv_id":"2302.02788","repositories_listed":0,"syntology":null},{"url":null,"slug":"refined-value-based-offline-rl-under","title":"Offline Minimax Soft-Q-learning Under Realizability and Partial Coverage","date":"2023-02-05","arxiv_id":"2302.02392","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-uncertainty-propagation-in-offline","title":"Selective Uncertainty Propagation in Offline RL","date":"2023-02-01","arxiv_id":"2302.00284","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-decision-transformer","title":"Skill Decision Transformer","date":"2023-01-31","arxiv_id":"2301.13573","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-view-decision-transformers-for","title":"Learning to View: Decision Transformers for Active Object Detection","date":"2023-01-23","arxiv_id":"2301.09544","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarks-and-algorithms-for-offline","title":"Benchmarks and Algorithms for Offline Preference-Based Reward Learning","date":"2023-01-03","arxiv_id":"2301.01392","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-evaluation-for-reinforcement-learning","title":"Offline Evaluation for Reinforcement Learning-based Recommendation: A Critical Issue and Some Alternatives","date":"2023-01-03","arxiv_id":"2301.00993","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-policy-optimization-in-rl-with","title":"Offline Policy Optimization in RL with Variance Regularizaton","date":"2022-12-29","arxiv_id":"2212.14405","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-in-deep-rl-via","title":"Representation Learning in Deep RL via Discrete Information Bottleneck","date":"2022-12-28","arxiv_id":"2212.13835","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-linear-programming-framework","title":"Offline Reinforcement Learning via Linear-Programming with Error-Bound Induced Constraints","date":"2022-12-28","arxiv_id":"2212.13861","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-offline-and-online","title":"Bridging the Gap Between Offline and Online Reinforcement Learning Evaluation Methodologies","date":"2022-12-15","arxiv_id":"2212.08131","repositories_listed":0,"syntology":null},{"url":null,"slug":"confidence-conditioned-value-functions-for","title":"Confidence-Conditioned Value Functions for Offline Reinforcement Learning","date":"2022-12-08","arxiv_id":"2212.04607","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-offline-reinforcement-learning","title":"Benchmarking Offline Reinforcement Learning Algorithms for E-Commerce Order Fraud Evaluation","date":"2022-12-05","arxiv_id":"2212.02620","repositories_listed":0,"syntology":null},{"url":null,"slug":"launchpad-learning-to-schedule-using-offline","title":"Launchpad: Learning to Schedule Using Offline and Online RL Methods","date":"2022-12-01","arxiv_id":"2212.00639","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-policy-evaluation-and-optimization","title":"Offline Policy Evaluation and Optimization under Confounding","date":"2022-11-29","arxiv_id":"2211.16583","repositories_listed":0,"syntology":null}],"record_sha256":"63868f8c4968b3f063963930d5db1e12bbe94be460642107cc3ba5f7350e8395","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}