{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/stereoset-measuring-stereotypical-bias-in","title":"StereoSet: Measuring stereotypical bias in pretrained language models","arxiv_id":"2004.09456","date":"2020-04-20","proceeding":"ACL 2021 5","authors":["Moin Nadeem","Anna Bethke","Siva Reddy"],"abstract":"A stereotype is an over-generalized belief about a particular group of people, e.g., Asians are good at math or Asians are bad drivers. Such beliefs (biases) are known to hurt target groups. Since pretrained language models are trained on large real world data, they are known to capture stereotypical biases. In order to assess the adverse effects of these models, it is important to quantify the bias captured in them. Existing literature on quantifying bias evaluates pretrained language models on a small set of artificially constructed bias-assessing sentences. We present StereoSet, a large-scale natural dataset in English to measure stereotypical biases in four domains: gender, profession, race, and religion. We evaluate popular models like BERT, GPT-2, RoBERTa, and XLNet on our dataset and show that these models exhibit strong stereotypical biases. We also present a leaderboard with a hidden test set to track the bias of future language models at https://stereoset.mit.edu","url_abs":"https://arxiv.org/abs/2004.09456v1","url_pdf":"https://arxiv.org/pdf/2004.09456v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"stereoset-measuring-stereotypical-bias-in","repo_url":"https://github.com/moinnadeem/StereoSet","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"CC-BY-SA-4.0"}},{"paper_slug":"stereoset-measuring-stereotypical-bias-in","repo_url":"https://github.com/kanekomasahiro/evaluate_bias_in_mlm","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"stereoset-measuring-stereotypical-bias-in","repo_url":"https://github.com/zalkikar/mlm-bias","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"bias-detection","task_name":"Bias Detection"},{"task_slug":"math","task_name":"Math"}],"methods":[{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"attention-dropout","method_name":"Attention Dropout"},{"method_slug":"bert","method_name":"BERT"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"cosine-annealing","method_name":"Cosine Annealing"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"discriminative-fine-tuning","method_name":"Discriminative Fine-Tuning"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"gpt-2","method_name":"GPT-2"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"linear-warmup-with-cosine-annealing","method_name":"Linear Warmup With Cosine Annealing"},{"method_slug":"linear-warmup-with-linear-decay","method_name":"Linear Warmup With Linear Decay"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"roberta","method_name":"RoBERTa"},{"method_slug":"sentencepiece","method_name":"SentencePiece"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"weight-decay","method_name":"Weight Decay"},{"method_slug":"wordpiece","method_name":"WordPiece"},{"method_slug":"xlnet","method_name":"XLNet"}],"datasets_introduced":[{"slug":"stereoset","name":"StereoSet","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"GPT-2 (small)","rank_in_archive_order":1,"of":11,"metrics":{"ICAT Score":"72.97"},"uses_additional_data":false},{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"XLNet (large)","rank_in_archive_order":2,"of":11,"metrics":{"ICAT Score":"72.03"},"uses_additional_data":false},{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"GPT-2 (medium)","rank_in_archive_order":3,"of":11,"metrics":{"ICAT Score":"71.73"},"uses_additional_data":false},{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"BERT (base)","rank_in_archive_order":4,"of":11,"metrics":{"ICAT Score":"71.21"},"uses_additional_data":false},{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"GPT-2 (large)","rank_in_archive_order":5,"of":11,"metrics":{"ICAT Score":"70.54"},"uses_additional_data":false},{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"BERT (large)","rank_in_archive_order":6,"of":11,"metrics":{"ICAT Score":"69.89"},"uses_additional_data":false},{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"RoBERTa (base)","rank_in_archive_order":7,"of":11,"metrics":{"ICAT Score":"67.50"},"uses_additional_data":false},{"leaderboard":"/sota/bias-detection-on-stereoset-1","task":"Bias Detection","dataset":"StereoSet","model":"XLNet (base)","rank_in_archive_order":9,"of":11,"metrics":{"ICAT Score":"62.10"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2004.09456","atlas_url":"https://app.syntology.ai/?focus=2004.09456","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}