{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/detecting-critical-treatment-effect-bias-in","title":"Detecting critical treatment effect bias in small subgroups","arxiv_id":"2404.18905","date":"2024-04-29","proceeding":null,"authors":["Piersilvio De Bartolomeis","Javier Abad","Konstantin Donhauser","Fanny Yang"],"abstract":"Randomized trials are considered the gold standard for making informed decisions in medicine, yet they often lack generalizability to the patient populations in clinical practice. Observational studies, on the other hand, cover a broader patient population but are prone to various biases. Thus, before using an observational study for decision-making, it is crucial to benchmark its treatment effect estimates against those derived from a randomized trial. We propose a novel strategy to benchmark observational studies beyond the average treatment effect. First, we design a statistical test for the null hypothesis that the treatment effects estimated from the two studies, conditioned on a set of relevant features, differ up to some tolerance. We then estimate an asymptotically valid lower bound on the maximum bias strength for any subgroup in the observational study. Finally, we validate our benchmarking strategy in a real-world setting and show that it leads to conclusions that align with established medical knowledge.","url_abs":"https://arxiv.org/abs/2404.18905v2","url_pdf":"https://arxiv.org/pdf/2404.18905v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"detecting-critical-treatment-effect-bias-in","repo_url":"https://github.com/jaabmar/kernel-test-bias","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"jax","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"benchmarking","task_name":"Benchmarking"},{"task_slug":"decision-making","task_name":"Decision Making"},{"task_slug":null,"task_name":"valid"}],"methods":[{"method_slug":"align","method_name":"ALIGN"},{"method_slug":"set","method_name":"SET"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=2404.18905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18905"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jaabmar/kernel-test-bias","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":6},"by_repo_kind":{"official":{"samples":6,"ran":6,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"61d7ae5359776aa3","entry":"add_bias_scenario_1","repo":"jaabmar/kernel-test-bias","repo_kind":"official","path":"src/datasets/bias_models.py","file_url":"https://github.com/jaabmar/kernel-test-bias/blob/HEAD/src/datasets/bias_models.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"61d7ae5359776aa3"}},{"code_sha256_prefix":"efa71661feab89d6","entry":"add_bias_scenario_2","repo":"jaabmar/kernel-test-bias","repo_kind":"official","path":"src/datasets/bias_models.py","file_url":"https://github.com/jaabmar/kernel-test-bias/blob/HEAD/src/datasets/bias_models.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"efa71661feab89d6"}},{"code_sha256_prefix":"27f35709e54e53e9","entry":"add_bias_subgroups","repo":"jaabmar/kernel-test-bias","repo_kind":"official","path":"src/datasets/bias_models.py","file_url":"https://github.com/jaabmar/kernel-test-bias/blob/HEAD/src/datasets/bias_models.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"27f35709e54e53e9"}},{"code_sha256_prefix":"70402deedf9f5526","entry":"estimate_selection_score","repo":"jaabmar/kernel-test-bias","repo_kind":"official","path":"src/experiment_utils.py","file_url":"https://github.com/jaabmar/kernel-test-bias/blob/HEAD/src/experiment_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"70402deedf9f5526"}},{"code_sha256_prefix":"ea617f3766473da5","entry":"initialize_model","repo":"jaabmar/kernel-test-bias","repo_kind":"official","path":"src/experiment_utils.py","file_url":"https://github.com/jaabmar/kernel-test-bias/blob/HEAD/src/experiment_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ea617f3766473da5"}},{"code_sha256_prefix":"f7d83fb9da13d37e","entry":"split_dataset","repo":"jaabmar/kernel-test-bias","repo_kind":"official","path":"src/datasets/hillstrom.py","file_url":"https://github.com/jaabmar/kernel-test-bias/blob/HEAD/src/datasets/hillstrom.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f7d83fb9da13d37e"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}