{"url":"/dataset/crows-pairs","name":"CrowS-Pairs","full_name":null,"description_markdown":"CrowS-Pairs has 1508 examples that cover stereotypes dealing with nine types of bias, like race, religion, and age. In CrowS-Pairs a model is presented with two sentences: one that is more stereotyping and another that is less stereotyping. The data focuses on stereotypes about historically disadvantaged groups and contrasts them with advantaged groups.\r\n\r\nSource: [CrowS-Pairs: A Challenge Dataset for Measuring Social Biases in Masked Language Models](/paper/crows-pairs-a-challenge-dataset-for-measuring)","description_withheld":null,"homepage":"https://github.com/nyu-mll/crows-pairs","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/crows-pairs-a-challenge-dataset-for-measuring","title":"CrowS-Pairs: A Challenge Dataset for Measuring Social Biases in Masked Language Models","first_author":"Nikita Nangia","url":null},"license":null,"modalities":[],"tasks":[{"name":"Stereotypical Bias Analysis","url":"/task/stereotypical-bias-analysis","datasets_with_task":"/datasets/task/stereotypical-bias-analysis"}],"languages":[],"variants":["CrowS-Pairs"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/nyu-mll/crows_pairs","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/crows_pairs","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/nyu-mll/crows-pairs","url":"https://github.com/nyu-mll/crows-pairs","frameworks":["pytorch"]}],"num_papers_in_archive":147,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/stereotypical-bias-analysis-on-crows-pairs","task":"Stereotypical Bias Analysis","dataset_variant":"CrowS-Pairs","rows":4,"metrics":["Gender","Religion","Race/Color","Sexual Orientation","Age","Nationality","Disability","Physical Appearance","Socioeconomic status","Overall"],"first_row_in_archive_order":{"model":"GAL 120B","paper":"/paper/galactica-a-large-language-model-for-science-1","metrics":{"Age":"69","Disability":"66.7","Gender":"51.9","Nationality":"51.6","Overall":"60.5","Physical Appearance":"58.7","Race/Color":"59.9","Religion":"51.9","Sexual Orientation":"77.4","Socioeconomic status":"65.7"},"code_links":[{"title":"paperswithcode/galai","url":"https://github.com/paperswithcode/galai"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":1,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/opt-open-pre-trained-transformer-language","title":"OPT: Open Pre-trained Transformer Language Models","date":"2022-05-02","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":6,"samples_unverified":18,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":84,"samples_ran":32,"samples_unverified":52,"pointer_only_for_licence":20,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}