{"url":"/task/off-policy-evaluation","name":"Off-policy evaluation","slug":"off-policy-evaluation","description_markdown":"Off-policy Evaluation (OPE), or offline evaluation in general, evaluates the performance of hypothetical policies leveraging only offline log data. It is particularly useful in applications where the online interaction involves high stakes and expensive setting such as precision medicine and recommender systems.","categories":[{"name":"Computer Code","url":"/area/computer-code"},{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Knowledge Base","url":"/area/knowledge-base"},{"name":"Methodology","url":"/area/methodology"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":265,"papers_with_code":102,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[{"url":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":102,"tagged_in_all":265,"items":[{"url":"/paper/a-large-scale-open-dataset-for-bandit","title":"Open Bandit Dataset and Pipeline: Towards Realistic and Reproducible Off-Policy Evaluation","date":"2020-08-17","arxiv_id":"2008.07146","repositories_listed":4,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/off-policy-evaluation-for-large-action-spaces","title":"Off-Policy Evaluation for Large Action Spaces via Embeddings","date":"2022-02-13","arxiv_id":"2202.06317","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/benchmarks-for-deep-off-policy-evaluation-1","title":"Benchmarks for Deep Off-Policy Evaluation","date":"2021-03-30","arxiv_id":"2103.16596","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/optimal-and-adaptive-off-policy-evaluation-in","title":"Optimal and Adaptive Off-policy Evaluation in Contextual Bandits","date":"2016-12-04","arxiv_id":"1612.01205","repositories_listed":3,"syntology":null},{"url":"/paper/balanced-off-policy-evaluation-for","title":"Balanced Off-Policy Evaluation for Personalized Pricing","date":"2023-02-24","arxiv_id":"2302.12736","repositories_listed":2,"syntology":null},{"url":"/paper/doubly-robust-off-policy-evaluation-for","title":"Doubly Robust Off-Policy Evaluation for Ranking Policies under the Cascade Behavior Model","date":"2022-02-03","arxiv_id":"2202.01562","repositories_listed":2,"syntology":null},{"url":"/paper/evaluating-the-robustness-of-off-policy","title":"Evaluating the Robustness of Off-Policy Evaluation","date":"2021-08-31","arxiv_id":"2108.13703","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/robust-generalization-despite-distribution","title":"Robust Generalization despite Distribution Shift via Minimum Discriminating Information","date":"2021-06-08","arxiv_id":"2106.04443","repositories_listed":2,"syntology":null},{"url":"/paper/confident-off-policy-evaluation-and-selection","title":"Confident Off-Policy Evaluation and Selection through Self-Normalized Importance Weighting","date":"2020-06-18","arxiv_id":"2006.10460","repositories_listed":2,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/intrinsically-efficient-stable-and-bounded","title":"Intrinsically Efficient, Stable, and Bounded Off-Policy Evaluation for Reinforcement Learning","date":"2019-06-09","arxiv_id":"1906.03735","repositories_listed":2,"syntology":null},{"url":"/paper/dolce-decomposing-off-policy-evaluation","title":"DOLCE: Decomposing Off-Policy Evaluation/Learning into Lagged and Current Effects","date":"2025-05-02","arxiv_id":"2505.00961","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-world-models-for-heterogeneous","title":"Trajectory World Models for Heterogeneous Environments","date":"2025-02-03","arxiv_id":"2502.01366","repositories_listed":1,"syntology":null},{"url":"/paper/two-way-deconfounder-for-off-policy","title":"Two-way Deconfounder for Off-policy Evaluation in Causal Reinforcement Learning","date":"2024-12-08","arxiv_id":"2412.05783","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":4}},{"url":"/paper/minimum-empirical-divergence-for-sub-gaussian","title":"Minimum Empirical Divergence for Sub-Gaussian Linear Bandits","date":"2024-10-31","arxiv_id":"2411.00229","repositories_listed":1,"syntology":null},{"url":"/paper/abstract-reward-processes-leveraging-state","title":"Abstract Reward Processes: Leveraging State Abstraction for Consistent Off-Policy Evaluation","date":"2024-10-03","arxiv_id":"2410.02172","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/causal-deepsets-for-off-policy-evaluation","title":"Causal Deepsets for Off-policy Evaluation under Spatial or Spatio-temporal Interferences","date":"2024-07-25","arxiv_id":"2407.17910","repositories_listed":1,"syntology":null},{"url":"/paper/forward-and-backward-state-abstractions-for","title":"Off-policy Evaluation with Deeply-abstracted States","date":"2024-06-27","arxiv_id":"2406.19531","repositories_listed":1,"syntology":null},{"url":"/paper/kernel-metric-learning-for-in-sample-off","title":"Kernel Metric Learning for In-Sample Off-Policy Evaluation of Deterministic RL Policies","date":"2024-05-29","arxiv_id":"2405.18792","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/cross-validated-off-policy-evaluation","title":"Cross-Validated Off-Policy Evaluation","date":"2024-05-24","arxiv_id":"2405.15332","repositories_listed":1,"syntology":null},{"url":"/paper/logarithmic-smoothing-for-pessimistic-off","title":"Logarithmic Smoothing for Pessimistic Off-Policy Evaluation, Selection and Learning","date":"2024-05-23","arxiv_id":"2405.14335","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/long-term-off-policy-evaluation-and-learning","title":"Long-term Off-Policy Evaluation and Learning","date":"2024-04-24","arxiv_id":"2404.15691","repositories_listed":1,"syntology":null},{"url":"/paper/predictive-performance-comparison-of-decision","title":"Predictive Performance Comparison of Decision Policies Under Confounding","date":"2024-04-01","arxiv_id":"2404.00848","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/efficient-and-sharp-off-policy-evaluation-in","title":"Efficient and Sharp Off-Policy Evaluation in Robust Markov Decision Processes","date":"2024-03-29","arxiv_id":"2404.00099","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":3}},{"url":"/paper/off-policy-evaluation-of-slate-bandit","title":"Off-Policy Evaluation of Slate Bandit Policies via Optimizing Abstraction","date":"2024-02-03","arxiv_id":"2402.02171","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-off-policy-evaluation-with","title":"Distributional Off-policy Evaluation with Bellman Residual Minimization","date":"2024-02-02","arxiv_id":"2402.01900","repositories_listed":1,"syntology":null},{"url":"/paper/debiased-machine-learning-and-network","title":"RoME: A Robust Mixed-Effects Bandit Algorithm for Optimizing Mobile Health Interventions","date":"2023-12-11","arxiv_id":"2312.06403","repositories_listed":1,"syntology":null},{"url":"/paper/marginal-density-ratio-for-off-policy-1","title":"Marginal Density Ratio for Off-Policy Evaluation in Contextual Bandits","date":"2023-12-03","arxiv_id":"2312.01457","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/towards-assessing-and-benchmarking-risk","title":"Towards Assessing and Benchmarking Risk-Return Tradeoff of Off-Policy Evaluation","date":"2023-11-30","arxiv_id":"2311.18207","repositories_listed":1,"syntology":null},{"url":"/paper/scope-rl-a-python-library-for-offline","title":"SCOPE-RL: A Python Library for Offline Reinforcement Learning and Off-Policy Evaluation","date":"2023-11-30","arxiv_id":"2311.18206","repositories_listed":1,"syntology":null},{"url":"/paper/when-is-off-policy-evaluation-useful-a-data","title":"When is Off-Policy Evaluation (Reward Modeling) Useful in Contextual Bandits? A Data-Centric Perspective","date":"2023-11-23","arxiv_id":"2311.14110","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}