{"url":"/dataset/winoground","name":"Winoground","full_name":null,"description_markdown":"Winoground is a dataset for evaluating the ability of vision and language models to conduct visio-linguistic compositional reasoning. Given two images and two captions, the goal is to match them correctly -- but crucially, both captions contain a completely identical set of words, only in a different order. The dataset was carefully hand-curated by expert annotators and is labeled with a rich set of fine-grained tags to assist in analyzing model performance.","description_withheld":null,"homepage":"https://huggingface.co/datasets/facebook/winoground","introduced_date":"2022-04-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/winoground-probing-vision-and-language-models","title":"Winoground: Probing Vision and Language Models for Visio-Linguistic Compositionality","first_author":"Tristan Thrush","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Visual Reasoning","url":"/task/visual-reasoning","datasets_with_task":"/datasets/task/visual-reasoning"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Winoground"],"data_loaders":[],"num_papers_in_archive":90,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-reasoning-on-winoground","task":"Visual Reasoning","dataset_variant":"Winoground","rows":114,"metrics":["Text Score","Image Score","Group Score"],"first_row_in_archive_order":{"model":"GPT-4o + CA","paper":"/paper/cognitive-paradigms-for-evaluating-vlms-on","metrics":{"Group Score":"52","Image Score":"58.5","Text Score":"75.5"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cognitive-paradigms-for-evaluating-vlms-on","title":"A Cognitive Paradigm Approach to Probe the Perception-Reasoning Interface in VLMs","date":"2025-01-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/prompting-large-vision-language-models-for","title":"Prompting Large Vision-Language Models for Compositional Reasoning","date":"2024-01-20","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cocot-contrastive-chain-of-thought-prompting","title":"CoCoT: Contrastive Chain-of-Thought Prompting for Large Multimodal Models with Multiple Image Inputs","date":"2024-01-05","rows_on_this_dataset":13,"code_links":1,"syntology":null},{"paper":"/paper/compositional-chain-of-thought-prompting-for","title":"Compositional Chain-of-Thought Prompting for Large Multimodal Models","date":"2023-11-27","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/selfeval-leveraging-the-discriminative-nature","title":"SelfEval: Leveraging the discriminative nature of generative models for evaluation","date":"2023-11-17","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/the-role-of-chain-of-thought-in-complex","title":"The Role of Chain-of-Thought in Complex Vision-Language Reasoning Task","date":"2023-11-15","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/mmicl-empowering-vision-language-model-with","title":"MMICL: Empowering Vision-language Model with Multi-Modal In-Context Learning","date":"2023-09-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-examination-of-the-compositionality-of","title":"An Examination of the Compositionality of Large Generative Vision-Language Models","date":"2023-08-21","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/visualgptscore-visio-linguistic-reasoning","title":"Revisiting the Role of Language Priors in Vision-Language Models","date":"2023-06-02","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/what-you-see-is-what-you-read-improving-text-1","title":"What You See is What You Read? Improving Text-Image Alignment Evaluation","date":"2023-05-17","rows_on_this_dataset":8,"code_links":1,"syntology":null},{"paper":"/paper/measuring-progress-in-fine-grained-vision-and","title":"Measuring Progress in Fine-grained Vision-and-Language Understanding","date":"2023-05-12","rows_on_this_dataset":9,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-token-level-confidence-improves","title":"Simple Token-Level Confidence Improves Caption Correctness","date":"2023-05-11","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/incorporating-structured-representations-into","title":"Incorporating Structured Representations into Pretrained Vision & Language Models Using Scene Graphs","date":"2023-05-10","rows_on_this_dataset":12,"code_links":0,"syntology":null},{"paper":"/paper/going-beyond-nouns-with-vision-language","title":"Going Beyond Nouns With Vision & Language Models Using Synthetic Data","date":"2023-03-30","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/your-diffusion-model-is-secretly-a-zero-shot","title":"Your Diffusion Model is Secretly a Zero-Shot Classifier","date":"2023-03-28","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/equivariant-similarity-for-vision-language","title":"Equivariant Similarity for Vision-Language Foundation Models","date":"2023-03-25","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vilem-visual-language-error-modeling-for","title":"ViLEM: Visual-Language Error Modeling for Image-Text Retrieval","date":"2023-01-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/does-structural-attention-improve","title":"Does Structural Attention Improve Compositional Representations in Vision-Language Models?","date":"2022-12-03","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/winoground-probing-vision-and-language-models","title":"Winoground: Probing Vision and Language Models for Visio-Linguistic Compositionality","date":"2022-04-07","rows_on_this_dataset":20,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":24,"samples_ran":17,"samples_unverified":7,"pointer_only_for_licence":11,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}