{"url":"/dataset/wsc","name":"WSC","full_name":"Winograd Schema Challenge","description_markdown":"The **Winograd Schema Challenge** was introduced both as an alternative to the Turing Test and as a test of a system’s ability to do commonsense reasoning. A Winograd schema is a pair of sentences differing in one or two words with a highly ambiguous pronoun, resolved differently in the two sentences, that appears to require commonsense knowledge to be resolved correctly. The examples were designed to be easily solvable by humans but difficult for machines, in principle requiring a deep understanding of the content of the text and the situation it describes.\r\n\r\nThe original Winograd Schema Challenge dataset consisted of 100 Winograd schemas constructed manually by AI experts. As of 2020 there are 285 examples available; however, the last 12 examples were only added recently. To ensure consistency with earlier models, several authors often prefer to report the performance on the first 273 examples only. These datasets are usually referred to as **WSC**285 and WSC273, respectively.\r\n\r\nSource: [https://arxiv.org/pdf/2004.13831.pdf](https://arxiv.org/pdf/2004.13831.pdf)\r\nImage Source: [https://arxiv.org/pdf/1907.11983.pdf](https://arxiv.org/pdf/1907.11983.pdf)","description_withheld":null,"homepage":"https://cs.nyu.edu/faculty/davise/papers/WinogradSchemas/WS.html","introduced_date":"2012-01-01","introduced_date_note":null,"introduced_by":{"paper":null,"title":"The Winograd Schema Challenge","first_author":null,"url":"http://www.aaai.org/ocs/index.php/KR/KR12/paper/view/4492"},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Classification","url":"/task/classification-1","datasets_with_task":"/datasets/task/classification-1"},{"name":"Coreference Resolution","url":"/task/coreference-resolution","datasets_with_task":"/datasets/task/coreference-resolution"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Winograd Schema Challenge","WSC","winograd_wsc"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/winograd_wsc","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/coref-data/winograd_wsc_raw","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/coref-data/winograd_wsc","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/coref-data/davis_wsc_raw","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ErnestSDavis/winograd_wsc","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/wsc273","frameworks":["tf","jax"]}],"num_papers_in_archive":361,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/coreference-resolution-on-winograd-schema","task":"Coreference Resolution","dataset_variant":"Winograd Schema Challenge","rows":82,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 540B (fine-tuned)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Accuracy":"100"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/classification-on-wsc","task":"Classification","dataset_variant":"WSC","rows":2,"metrics":["Test Accuracy"],"first_row_in_archive_order":{"model":"OPT-1.3B","paper":"/paper/achieving-dimension-free-communication-in","metrics":{"Test Accuracy":"64.16%"},"code_links":[{"title":"ZidongLiu/DeComFL","url":"https://github.com/ZidongLiu/DeComFL"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/achieving-dimension-free-communication-in","title":"Achieving Dimension-Free Communication in Federated Learning via Zeroth-Order Optimization","date":"2024-05-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-cot-collection-improving-zero-shot-and","title":"The CoT Collection: Improving Zero-shot and Few-shot Learning of Language Models via Chain-of-Thought Fine-Tuning","date":"2023-05-23","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/lamini-lm-a-diverse-herd-of-distilled-models","title":"LaMini-LM: A Diverse Herd of Distilled Models from Large-Scale Instructions","date":"2023-04-27","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/pythia-a-suite-for-analyzing-large-language","title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","date":"2023-04-03","rows_on_this_dataset":4,"code_links":4,"syntology":null},{"paper":"/paper/exploring-the-benefits-of-training-expert","title":"Exploring the Benefits of Training Expert Language Models over Instruction Tuning","date":"2023-02-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hungry-hungry-hippos-towards-language","title":"Hungry Hungry Hippos: Towards Language Modeling with State Space Models","date":"2022-12-28","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":7,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/toward-efficient-language-model-pretraining","title":"Toward Efficient Language Model Pretraining and Downstream Adaptation via Self-Evolution: A Case Study on SuperGLUE","date":"2022-12-04","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/knowledge-in-context-towards-knowledgeable","title":"Knowledge-in-Context: Towards Knowledgeable Semi-Parametric Language Models","date":"2022-10-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scaling-instruction-finetuned-language-models","title":"Scaling Instruction-Finetuned Language Models","date":"2022-10-20","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":8,"samples_unverified":9,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/guess-the-instruction-making-language-models","title":"Guess the Instruction! Flipped Learning Makes Language Models Stronger Zero-Shot Learners","date":"2022-10-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ask-me-anything-a-simple-strategy-for","title":"Ask Me Anything: A simple strategy for prompting language models","date":"2022-10-05","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/alexatm-20b-few-shot-learning-using-a-large","title":"AlexaTM 20B: Few-Shot Learning Using a Large-Scale Multilingual Seq2Seq Model","date":"2022-08-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/n-grammer-augmenting-transformers-with-latent-1","title":"N-Grammer: Augmenting Transformers with latent n-grams","date":"2022-07-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":0,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":4,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":30,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/designing-effective-sparse-expert-models","title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","date":"2022-02-17","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-generalization-in-coreference-resolution","title":"On Generalization in Coreference Resolution","date":"2021-09-20","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/back-to-square-one-bias-detection-training","title":"Back to Square One: Artifact Detection, Training and Commonsense Disentanglement in the Winograd Schema","date":"2021-04-16","rows_on_this_dataset":7,"code_links":0,"syntology":null},{"paper":"/paper/deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":4,"samples_unverified":9,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/g-daug-generative-data-augmentation-for","title":"Generative Data Augmentation for Commonsense Reasoning","date":"2020-04-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tttttackling-winogrande-schemas","title":"TTTTTackling WinoGrande Schemas","date":"2020-03-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","rows_on_this_dataset":1,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":2,"samples_unverified":29,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-hybrid-neural-network-model-for-commonsense","title":"A Hybrid Neural Network Model for Commonsense Reasoning","date":"2019-07-27","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/winogrande-an-adversarial-winograd-schema","title":"WinoGrande: An Adversarial Winograd Schema Challenge at Scale","date":"2019-07-24","rows_on_this_dataset":4,"code_links":10,"syntology":null},{"paper":"/paper/attention-is-not-all-you-need-for-commonsense","title":"Attention Is (not) All You Need for Commonsense Reasoning","date":"2019-05-31","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/a-surprisingly-robust-trick-for-winograd","title":"A Surprisingly Robust Trick for Winograd Schema Challenge","date":"2019-05-15","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/socialiqa-commonsense-reasoning-about-social","title":"SocialIQA: Commonsense Reasoning about Social Interactions","date":"2019-04-22","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unsupervised-deep-structured-semantic-models","title":"Unsupervised Deep Structured Semantic Models for Commonsense Reasoning","date":"2019-04-03","rows_on_this_dataset":5,"code_links":0,"syntology":null},{"paper":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","rows_on_this_dataset":1,"code_links":21,"syntology":null},{"paper":"/paper/on-the-evaluation-of-common-sense-reasoning","title":"How Reasonable are Common-Sense Reasoning Tasks: A Case-Study on the Winograd Schema Challenge and SWAG","date":"2018-11-05","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","rows_on_this_dataset":1,"code_links":534,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":659,"samples_ran":204,"samples_unverified":455,"pointer_only_for_licence":149,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-knowledge-hunting-framework-for-common","title":"A Knowledge Hunting Framework for Common Sense Reasoning","date":"2018-10-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-simple-method-for-commonsense-reasoning","title":"A Simple Method for Commonsense Reasoning","date":"2018-06-07","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","rows_on_this_dataset":1,"code_links":595,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":946,"samples_ran":600,"samples_unverified":346,"pointer_only_for_licence":451,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/commonsense-knowledge-enhanced-embeddings-for","title":"Commonsense Knowledge Enhanced Embeddings for Solving Pronoun Disambiguation Problems in Winograd Schema Challenge","date":"2016-11-13","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":20,"samples_harvested":1840,"samples_ran":884,"samples_unverified":956,"pointer_only_for_licence":621,"papers_with_no_sample_that_ran":6,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}