{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/offline-rl/papers/8","list_of":"/task/offline-rl","task":"Offline RL","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":8,"rows_per_page":100,"rows":[701,755],"of":755,"counts":{"archive_papers_tagged":755,"with_a_code_link":310,"where_syntology_ran_a_sample":164,"not_listed_spam_title":0,"listed":755,"listed_where_code_ran":164,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":139,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":139,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/offline-rl","prev":"/task/offline-rl/papers/7","next":null,"papers":[{"url":null,"slug":"why-so-pessimistic-estimating-uncertainties","title":"Why so pessimistic? Estimating uncertainties for offline RL through ensembles, and why their independence matters.","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-offline-reinforcement-learning","title":"Accelerating Offline Reinforcement Learning Application in Real-Time Bidding and Recommendation: Potential Use of Simulation","date":"2021-09-17","arxiv_id":"2109.08331","repositories_listed":0,"syntology":null},{"url":null,"slug":"conservative-data-sharing-for-multi-task","title":"Conservative Data Sharing for Multi-Task Offline Reinforcement Learning","date":"2021-09-16","arxiv_id":"2109.08128","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradients-incorporating-the-future","title":"Policy Gradients Incorporating the Future","date":"2021-08-04","arxiv_id":"2108.02096","repositories_listed":0,"syntology":null},{"url":null,"slug":"opal-offline-preference-based-apprenticeship","title":"Offline Preference-Based Apprenticeship Learning","date":"2021-07-20","arxiv_id":"2107.09251","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraints-penalized-q-learning-for-safe","title":"Constraints Penalized Q-learning for Safe Offline Reinforcement Learning","date":"2021-07-19","arxiv_id":"2107.09003","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-model-based-offline-rl-pac-bounds","title":"Pessimistic Model-based Offline Reinforcement Learning under Partial Coverage","date":"2021-07-13","arxiv_id":"2107.06226","repositories_listed":0,"syntology":null},{"url":null,"slug":"camtuner-reinforcement-learning-based-system","title":"Enhancing Video Analytics Accuracy via Real-time Automated Camera Parameter Tuning","date":"2021-07-08","arxiv_id":"2107.03964","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-least-restriction-for-offline","title":"The Least Restriction for Offline Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01757","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-representation-learning-in","title":"Provably Efficient Representation Selection in Low-rank Markov Decision Processes: From Online to Offline RL","date":"2021-06-22","arxiv_id":"2106.11935","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-offline-reinforcement-learning-with","title":"Boosting Offline Reinforcement Learning with Residual Generative Modeling","date":"2021-06-19","arxiv_id":"2106.10411","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioral-priors-and-dynamics-models","title":"Behavioral Priors and Dynamics Models: Improving Performance and Domain Transfer in Offline RL","date":"2021-06-16","arxiv_id":"2106.09119","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-multi-objective-policy-optimization-as-a","title":"On Multi-objective Policy Optimization as a Tool for Reinforcement Learning: Case Studies in Offline RL and Finetuning","date":"2021-06-15","arxiv_id":"2106.08199","repositories_listed":0,"syntology":null},{"url":null,"slug":"corruption-robust-offline-reinforcement","title":"Corruption-Robust Offline Reinforcement Learning","date":"2021-06-11","arxiv_id":"2106.06630","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-as-anti","title":"Offline Reinforcement Learning as Anti-Exploration","date":"2021-06-11","arxiv_id":"2106.06431","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-inverse-reinforcement-learning","title":"Offline Inverse Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.05068","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-finetuning-bridging-sample-efficient","title":"Policy Finetuning: Bridging Sample-Efficient Offline and Online Reinforcement Learning","date":"2021-06-09","arxiv_id":"2106.04895","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-long-term-metrics-in-recommendation","title":"Improving Long-Term Metrics in Recommendation Systems using Short-Horizon Reinforcement Learning","date":"2021-06-01","arxiv_id":"2106.00589","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-design-choices-in-offline-model-1","title":"Revisiting Design Choices in Offline Model Based Reinforcement Learning","date":"2021-05-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-uniform-convergence-in-offline","title":"Optimal Uniform OPE and Model-based Offline Reinforcement Learning in Time-Homogeneous, Reward-Free and Task-Agnostic Settings","date":"2021-05-13","arxiv_id":"2105.06029","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-performance-analysis-towards","title":"Interpretable performance analysis towards offline reinforcement learning: A dataset perspective","date":"2021-05-12","arxiv_id":"2105.05473","repositories_listed":0,"syntology":null},{"url":null,"slug":"infernet-for-delayed-reinforcement-tasks","title":"InferNet for Delayed Reinforcement Tasks: Addressing the Temporal Credit Assignment Problem","date":"2021-05-02","arxiv_id":"2105.00568","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-offline-reinforcement-learning-and","title":"Bridging Offline Reinforcement Learning and Imitation Learning: A Tale of Pessimism","date":"2021-03-22","arxiv_id":"2103.12021","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-behavior-value-estimation","title":"Regularized Behavior Value Estimation","date":"2021-03-17","arxiv_id":"2103.09575","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-finite-sample-analysis-of-offline","title":"Sample Complexity of Offline Reinforcement Learning with Deep ReLU Networks","date":"2021-03-11","arxiv_id":"2103.06671","repositories_listed":0,"syntology":null},{"url":null,"slug":"s4rl-surprisingly-simple-self-supervision-for","title":"S4RL: Surprisingly Simple Self-Supervision for Offline Reinforcement Learning","date":"2021-03-10","arxiv_id":"2103.06326","repositories_listed":0,"syntology":null},{"url":null,"slug":"instabilities-of-offline-rl-with-pre-trained","title":"Instabilities of Offline RL with Pre-Trained Neural Representation","date":"2021-03-08","arxiv_id":"2103.04947","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepthermal-combustion-optimization-for","title":"DeepThermal: Combustion Optimization for Thermal Power Generating Units Using Offline Reinforcement Learning","date":"2021-02-23","arxiv_id":"2102.11492","repositories_listed":0,"syntology":null},{"url":null,"slug":"gelato-geometrically-enriched-latent-model","title":"Uncertainty Estimation Using Riemannian Model~Dynamics for Offline Reinforcement Learning","date":"2021-02-22","arxiv_id":"2102.11327","repositories_listed":0,"syntology":null},{"url":null,"slug":"instrumental-variable-value-iteration-for","title":"Instrumental Variable Value Iteration for Causal Offline Reinforcement Learning","date":"2021-02-19","arxiv_id":"2102.09907","repositories_listed":0,"syntology":null},{"url":null,"slug":"persim-data-efficient-offline-reinforcement","title":"PerSim: Data-Efficient Offline Reinforcement Learning with Heterogeneous Agents via Personalized Simulators","date":"2021-02-13","arxiv_id":"2102.06961","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-matters-offline-pretraining","title":"Representation Matters: Offline Pretraining for Sequential Decision Making","date":"2021-02-11","arxiv_id":"2102.05815","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-offline-reinforcement-learning","title":"Near-Optimal Offline Reinforcement Learning via Double Variance Reduction","date":"2021-02-02","arxiv_id":"2102.01748","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-distribution-shift-in-online","title":"Addressing Distribution Shift in Online Reinforcement Learning with Offline Datasets","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-extrapolation-error-in-deep","title":"Addressing Extrapolation Error in Deep Offline Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"brac-going-deeper-with-behavior-regularized","title":"BRAC+: Going Deeper with Behavior Regularized Offline Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-policy-optimization-with-variance","title":"Offline Policy Optimization with Variance Regularization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-balancing-offline-model-based","title":"Representation Balancing Offline Model-based Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-offline-reinforcement-learning-from","title":"Robust Offline Reinforcement Learning from Low-Quality Data","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-weighted-offline-reinforcement","title":"Uncertainty Weighted Offline Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"is-pessimism-provably-efficient-for-offline","title":"Is Pessimism Provably Efficient for Offline RL?","date":"2020-12-30","arxiv_id":"2012.15085","repositories_listed":0,"syntology":null},{"url":null,"slug":"batch-constrained-distributional","title":"Batch-Constrained Distributional Reinforcement Learning for Session-based Recommendation","date":"2020-12-16","arxiv_id":"2012.08984","repositories_listed":0,"syntology":null},{"url":null,"slug":"morel-model-based-offline-reinforcement-1","title":"MOReL: Model-Based Offline Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-hands-on","title":"Offline Reinforcement Learning Hands-On","date":"2020-11-29","arxiv_id":"2011.14379","repositories_listed":0,"syntology":null},{"url":null,"slug":"opal-offline-primitive-discovery-for-1","title":"OPAL: Offline Primitive Discovery for Accelerating Offline Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-are-the-statistical-limits-of-offline-rl","title":"What are the Statistical Limits of Offline RL with Linear Function Approximation?","date":"2020-10-22","arxiv_id":"2010.11895","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dexterous-manipulation-from","title":"Learning Dexterous Manipulation from Suboptimal Experts","date":"2020-10-16","arxiv_id":"2010.08587","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-reinforcement-learning-based-multi-agent","title":"The reinforcement learning-based multi-agent cooperative approach for the adaptive speed regulation on a metallurgical pickling line","date":"2020-08-16","arxiv_id":"2008.06933","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-planning","title":"Model-Based Offline Planning","date":"2020-08-12","arxiv_id":"2008.05556","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-model-bias-for-robust-offline-deep","title":"Overcoming Model Bias for Robust Offline Deep Reinforcement Learning","date":"2020-08-12","arxiv_id":"2008.05533","repositories_listed":0,"syntology":null},{"url":null,"slug":"emaq-expected-max-q-learning-operator-for","title":"EMaQ: Expected-Max Q-Learning Operator for Simple Yet Effective Offline and Online RL","date":"2020-07-21","arxiv_id":"2007.11091","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperparameter-selection-for-offline","title":"Hyperparameter Selection for Offline Reinforcement Learning","date":"2020-07-17","arxiv_id":"2007.09055","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-provable-uniform-convergence-in","title":"Near-Optimal Provable Uniform Convergence in Offline Policy Evaluation for Reinforcement Learning","date":"2020-07-07","arxiv_id":"2007.03760","repositories_listed":0,"syntology":null},{"url":null,"slug":"striving-for-simplicity-in-off-policy-deep-1","title":"Striving for Simplicity in Off-Policy Deep Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-offline-goal-oriented-dialog","title":"End-to-End Offline Goal-Oriented Dialog Policy Learning via Policy Gradient","date":"2017-12-07","arxiv_id":"1712.02838","repositories_listed":0,"syntology":null}],"record_sha256":"c110882d259ba02b311b0b17dbe14b0aac698e8bf51dc09593cf6f74a51dc7fa","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}