{"url":"/task/selection-bias","name":"Selection bias","slug":"selection-bias","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":365,"papers_with_code":143,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/pannuke","name":"PanNuke","full_name":"","num_papers_in_archive":61},{"url":"/dataset/iclr-database","name":"ICLR Database","full_name":"ICLR Database (with Textual Covariates)","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/bias-detection","name":"Bias Detection"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":143,"tagged_in_all":365,"items":[{"url":"/paper/pannuke-dataset-extension-insights-and","title":"PanNuke Dataset Extension, Insights and Baselines","date":"2020-03-24","arxiv_id":"2003.10778","repositories_listed":8,"syntology":{"n":40,"n_ran":4,"n_unverified":36,"n_pointer_only":1}},{"url":"/paper/entire-space-multi-task-model-an-effective","title":"Entire Space Multi-Task Model: An Effective Approach for Estimating Post-Click Conversion Rate","date":"2018-04-21","arxiv_id":"1804.07931","repositories_listed":6,"syntology":null},{"url":"/paper/active-structure-learning-of-causal-dags-via","title":"Active Structure Learning of Causal DAGs via Directed Clique Tree","date":"2020-11-01","arxiv_id":"2011.00641","repositories_listed":4,"syntology":null},{"url":"/paper/a-debiased-mdi-feature-importance-measure-for","title":"A Debiased MDI Feature Importance Measure for Random Forests","date":"2019-06-26","arxiv_id":"1906.10845","repositories_listed":3,"syntology":null},{"url":"/paper/evos-efficient-implicit-neural-training-via","title":"EVOS: Efficient Implicit Neural Training via EVOlutionary Selector","date":"2024-12-13","arxiv_id":"2412.10153","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/investigating-subtler-biases-in-llms-ageism","title":"Investigating Subtler Biases in LLMs: Ageism, Beauty, Institutional, and Nationality Bias in Generative Models","date":"2023-09-16","arxiv_id":"2309.08902","repositories_listed":2,"syntology":null},{"url":"/paper/discovering-and-explaining-the-non-causality","title":"Discovering and Explaining the Non-Causality of Deep Learning in SAR ATR","date":"2023-04-03","arxiv_id":"2304.00668","repositories_listed":2,"syntology":null},{"url":"/paper/mathematical-capabilities-of-chatgpt-1","title":"Mathematical Capabilities of ChatGPT","date":"2023-01-31","arxiv_id":"2301.13867","repositories_listed":2,"syntology":null},{"url":"/paper/alleviating-the-sample-selection-bias-in-few","title":"Alleviating the Sample Selection Bias in Few-shot Learning by Removing Projection to the Centroid","date":"2022-10-30","arxiv_id":"2210.16834","repositories_listed":2,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":6}},{"url":"/paper/exploiting-selection-bias-on-underspecified","title":"Underspecification in Language Modeling Tasks: A Causality-Informed Study of Gendered Pronoun Resolution","date":"2022-09-30","arxiv_id":"2210.00131","repositories_listed":2,"syntology":null},{"url":"/paper/algorithm-is-experiment-machine-learning","title":"Algorithm as Experiment: Machine Learning, Market Design, and Policy Eligibility Rules","date":"2021-04-26","arxiv_id":"2104.12909","repositories_listed":2,"syntology":null},{"url":"/paper/unmasking-the-mask-evaluating-social-biases","title":"Unmasking the Mask -- Evaluating Social Biases in Masked Language Models","date":"2021-04-15","arxiv_id":"2104.07496","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/diagnostic-curves-for-black-box-models","title":"Automated Dependence Plots","date":"2019-12-02","arxiv_id":"1912.01108","repositories_listed":2,"syntology":null},{"url":"/paper/to-model-or-to-intervene-a-comparison-of","title":"To Model or to Intervene: A Comparison of Counterfactual and Online Learning to Rank from User Interactions","date":"2019-07-15","arxiv_id":"1907.06412","repositories_listed":2,"syntology":null},{"url":"/paper/selection-bias-explorations-and-debias","title":"Selection Bias Explorations and Debias Methods for Natural Language Sentence Matching Datasets","date":"2019-05-15","arxiv_id":"1905.06221","repositories_listed":2,"syntology":null},{"url":"/paper/adversarial-balancing-based-representation","title":"Adversarial Balancing-based Representation Learning for Causal Effect Inference with Observational Data","date":"2019-04-30","arxiv_id":"1904.13335","repositories_listed":2,"syntology":{"n":23,"n_ran":1,"n_unverified":22,"n_pointer_only":0}},{"url":"/paper/llm-generated-feedback-supports-learning-if","title":"LLM-Generated Feedback Supports Learning If Learners Choose to Use It","date":"2025-06-20","arxiv_id":"2506.17006","repositories_listed":1,"syntology":null},{"url":"/paper/addressing-correlated-latent-exogenous","title":"Addressing Correlated Latent Exogenous Variables in Debiased Recommender Systems","date":"2025-06-09","arxiv_id":"2506.07517","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-challenges-to-the","title":"Understanding challenges to the interpretation of disaggregated evaluations of algorithmic fairness","date":"2025-06-04","arxiv_id":"2506.04193","repositories_listed":1,"syntology":null},{"url":"/paper/representation-learning-preserving","title":"Representation Learning Preserving Ignorability and Covariate Matching for Treatment Effects","date":"2025-04-29","arxiv_id":"2504.20579","repositories_listed":1,"syntology":null},{"url":"/paper/negate-or-embrace-on-how-misalignment-shapes","title":"On the Value of Cross-Modal Misalignment in Multimodal Representation Learning","date":"2025-04-14","arxiv_id":"2504.10143","repositories_listed":1,"syntology":null},{"url":"/paper/when-selection-meets-intervention-additional","title":"When Selection Meets Intervention: Additional Complexities in Causal Discovery","date":"2025-03-10","arxiv_id":"2503.07302","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/metric-dst-mitigating-selection-bias-through","title":"Metric-DST: Mitigating Selection Bias Through Diversity-Guided Semi-Supervised Metric Learning","date":"2024-11-27","arxiv_id":"2411.18442","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-languages-are-equal-insights-into","title":"Not All Languages are Equal: Insights into Multilingual Retrieval-Augmented Generation","date":"2024-10-29","arxiv_id":"2410.21970","repositories_listed":1,"syntology":null},{"url":"/paper/recflow-an-industrial-full-flow","title":"RecFlow: An Industrial Full Flow Recommendation Dataset","date":"2024-10-28","arxiv_id":"2410.20868","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/heterogeneous-random-forest","title":"Heterogeneous Random Forest","date":"2024-10-24","arxiv_id":"2410.19022","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-coral-correlation-consistency","title":"Leveraging CORAL-Correlation Consistency Network for Semi-Supervised Left Atrium MRI Segmentation","date":"2024-10-21","arxiv_id":"2410.15916","repositories_listed":1,"syntology":null},{"url":"/paper/calibraeval-calibrating-prediction","title":"CalibraEval: Calibrating Prediction Distribution to Mitigate Selection Bias in LLMs-as-Judges","date":"2024-10-20","arxiv_id":"2410.15393","repositories_listed":1,"syntology":null},{"url":"/paper/diffpo-a-causal-diffusion-model-for-learning","title":"DiffPO: A causal diffusion model for learning distributions of potential outcomes","date":"2024-10-11","arxiv_id":"2410.08924","repositories_listed":1,"syntology":null},{"url":"/paper/dcast-diverse-class-aware-self-training","title":"DCAST: Diverse Class-Aware Self-Training Mitigates Selection Bias for Fairer Learning","date":"2024-09-30","arxiv_id":"2409.20126","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}