{"url":"/dataset/winogrande","name":"WinoGrande","full_name":null,"description_markdown":"WinoGrande is a large-scale dataset of 44k problems, inspired by the original WSC design, but adjusted to improve both the scale and the hardness of the dataset. The key steps of the dataset construction consist of (1) a carefully designed crowdsourcing procedure, followed by (2) systematic bias reduction using a novel AfLite algorithm that generalizes human-detectable word associations to machine-detectable embedding associations.\r\n\r\nSource: [WinoGrande: An Adversarial Winograd Schema Challenge at Scale](/paper/winogrande-an-adversarial-winograd-schema)\r\nImage Source: [https://winogrande.allenai.org/](https://winogrande.allenai.org/)","description_withheld":null,"homepage":"http://winogrande.allenai.org/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/winogrande-an-adversarial-winograd-schema","title":"WinoGrande: An Adversarial Winograd Schema Challenge at Scale","first_author":"Keisuke Sakaguchi","url":null},"license":{"name":"CC-BY","url":"https://github.com/allenai/winogrande"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"},{"name":"Common Sense Reasoning","url":"/task/common-sense-reasoning","datasets_with_task":"/datasets/task/common-sense-reasoning"},{"name":"Natural Language Understanding","url":"/task/natural-language-understanding","datasets_with_task":"/datasets/task/natural-language-understanding"},{"name":"parameter-efficient fine-tuning","url":"/task/parameter-efficient-fine-tuning","datasets_with_task":"/datasets/task/parameter-efficient-fine-tuning"},{"name":"Winogrande","url":"/task/winogrande","datasets_with_task":"/datasets/task/winogrande"}],"languages":[],"variants":["WinoGrande"," Winogrande","Winogrande (5-shot)","Winogrande TR v0.2","Winogrande TR"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/winogrande","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/coref-data/winogrande","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/coref-data/winogrande_raw","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/malhajar/winogrande-tr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/allenai/winogrande","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/winogrande","frameworks":["tf","jax"]}],"num_papers_in_archive":703,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/common-sense-reasoning-on-winogrande","task":"Common Sense Reasoning","dataset_variant":"WinoGrande","rows":77,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"ST-MoE-32B 269B (fine-tuned)","paper":"/paper/designing-effective-sparse-expert-models","metrics":{"Accuracy":"96.1"},"code_links":[{"title":"tensorflow/mesh","url":"https://github.com/tensorflow/mesh"},{"title":"xuefuzhao/openmoe","url":"https://github.com/xuefuzhao/openmoe"},{"title":"yikangshen/megablocks","url":"https://github.com/yikangshen/megablocks"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/parameter-efficient-fine-tuning-on-winogrande","task":"parameter-efficient fine-tuning","dataset_variant":"WinoGrande","rows":3,"metrics":["Accuracy (% )"],"first_row_in_archive_order":{"model":"LLaMA2-7b","paper":"/paper/gift-sw-gaussian-noise-injected-fine-tuning","metrics":{"Accuracy (% )":"70.80"},"code_links":[{"title":"On-Point-RND/GIFT_SW","url":"https://github.com/On-Point-RND/GIFT_SW"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-winogrande","task":"Text Generation","dataset_variant":"WinoGrande","rows":0,"metrics":["acc"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-winogrande-5-shot","task":"Text Generation","dataset_variant":"Winogrande (5-shot)","rows":0,"metrics":["accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-winogrande-tr","task":"Text Generation","dataset_variant":"Winogrande TR","rows":0,"metrics":["accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-winogrande-tr-v0-2","task":"Text Generation","dataset_variant":"Winogrande TR v0.2","rows":0,"metrics":["accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/winogrande-on-winogrande","task":"Winogrande","dataset_variant":"WinoGrande","rows":0,"metrics":["Accuracy"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/gift-sw-gaussian-noise-injected-fine-tuning","title":"GIFT-SW: Gaussian noise Injected Fine-Tuning of Salient Weights for LLMs","date":"2024-08-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixlora-enhancing-large-language-models-fine","title":"MixLoRA: Enhancing Large Language Models Fine-Tuning with LoRA-based Mixture of Experts","date":"2024-04-22","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-claude-3-model-family-opus-sonnet-haiku","title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","date":"2024-03-04","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/dora-weight-decomposed-low-rank-adaptation","title":"DoRA: Weight-Decomposed Low-Rank Adaptation","date":"2024-02-14","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":8,"samples_unverified":7,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/textbooks-are-all-you-need-ii-phi-1-5","title":"Textbooks Are All You Need II: phi-1.5 technical report","date":"2023-09-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-cot-collection-improving-zero-shot-and","title":"The CoT Collection: Improving Zero-shot and Few-shot Learning of Language Models via Chain-of-Thought Fine-Tuning","date":"2023-05-23","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/lamini-lm-a-diverse-herd-of-distilled-models","title":"LaMini-LM: A Diverse Herd of Distilled Models from Large-Scale Instructions","date":"2023-04-27","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/pythia-a-suite-for-analyzing-large-language","title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","date":"2023-04-03","rows_on_this_dataset":4,"code_links":4,"syntology":null},{"paper":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":4,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-the-benefits-of-training-expert","title":"Exploring the Benefits of Training Expert Language Models over Instruction Tuning","date":"2023-02-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/knowledge-in-context-towards-knowledgeable","title":"Knowledge-in-Context: Towards Knowledgeable Semi-Parametric Language Models","date":"2022-10-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/task-compass-scaling-multi-task-pre-training","title":"Task Compass: Scaling Multi-task Pre-training with Task Prefix","date":"2022-10-12","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/guess-the-instruction-making-language-models","title":"Guess the Instruction! Flipped Learning Makes Language Models Stronger Zero-Shot Learners","date":"2022-10-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":3,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":30,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-language-modeling-with-sparse-all","title":"Efficient Language Modeling with Sparse all-MLP","date":"2022-03-14","rows_on_this_dataset":5,"code_links":0,"syntology":null},{"paper":"/paper/designing-effective-sparse-expert-models","title":"ST-MoE: Designing Stable and Transferable Sparse Expert Models","date":"2022-02-17","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lora-low-rank-adaptation-of-large-language","title":"LoRA: Low-Rank Adaptation of Large Language Models","date":"2021-06-17","rows_on_this_dataset":1,"code_links":74,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":84,"samples_ran":34,"samples_unverified":50,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/back-to-square-one-bias-detection-training","title":"Back to Square One: Artifact Detection, Training and Commonsense Disentanglement in the Winograd Schema","date":"2021-04-16","rows_on_this_dataset":7,"code_links":0,"syntology":null},{"paper":"/paper/unicorn-on-rainbow-a-universal-commonsense","title":"UNICORN on RAINBOW: A Universal Commonsense Reasoning Model on a New Multitask Benchmark","date":"2021-03-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":2,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifiedqa-crossing-format-boundaries-with-a","title":"UnifiedQA: Crossing Format Boundaries With a Single QA System","date":"2020-05-02","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/g-daug-generative-data-augmentation-for","title":"Generative Data Augmentation for Commonsense Reasoning","date":"2020-04-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/winogrande-an-adversarial-winograd-schema","title":"WinoGrande: An Adversarial Winograd Schema Challenge at Scale","date":"2019-07-24","rows_on_this_dataset":6,"code_links":10,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":18,"samples_harvested":338,"samples_ran":164,"samples_unverified":174,"pointer_only_for_licence":78,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}