{"url":"/dataset/snli-ve","name":"SNLI-VE","full_name":null,"description_markdown":"Visual Entailment (VE) consists of image-sentence pairs whereby a premise is defined by an image, rather than a natural language sentence as in traditional Textual Entailment tasks. The goal of a trained VE model is to predict whether the image semantically entails the text. **SNLI-VE** is a dataset for VE which is based on the Stanford Natural Language Inference corpus and Flickr30k dataset.\n\nSource: [https://github.com/necla-ml/SNLI-VE](https://github.com/necla-ml/SNLI-VE)","description_withheld":null,"homepage":"https://github.com/necla-ml/SNLI-VE","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/visual-entailment-a-novel-task-for-fine","title":"Visual Entailment: A Novel Task for Fine-Grained Image Understanding","first_author":"Ning Xie","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"},{"name":"Visual Reasoning","url":"/task/visual-reasoning","datasets_with_task":"/datasets/task/visual-reasoning"},{"name":"Visual Entailment","url":"/task/visual-entailment","datasets_with_task":"/datasets/task/visual-entailment"}],"languages":[],"variants":["SNLI-VE val","SNLI-VE test","SNLI-VE"],"data_loaders":[{"repo":"https://github.com/allenai/allennlp-models","url":"https://docs.allennlp.org/models/main/models/vision/dataset_readers/visual_entailment/","frameworks":["pytorch"]},{"repo":"https://github.com/necla-ml/SNLI-VE","url":"https://github.com/necla-ml/SNLI-VE","frameworks":[]}],"num_papers_in_archive":117,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-entailment-on-snli-ve-val","task":"Visual Entailment","dataset_variant":"SNLI-VE val","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"OFA","paper":"/paper/unifying-architectures-tasks-and-modalities","metrics":{"Accuracy":"91.0"},"code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"ofa-sys/ofa","url":"https://github.com/ofa-sys/ofa"},{"title":"JHKim-snu/GVCCI","url":"https://github.com/JHKim-snu/GVCCI"},{"title":"JHKim-snu/PGA","url":"https://github.com/JHKim-snu/PGA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-entailment-on-snli-ve-test","task":"Visual Entailment","dataset_variant":"SNLI-VE test","rows":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"OFA","paper":"/paper/unifying-architectures-tasks-and-modalities","metrics":{"Accuracy":"91.2"},"code_links":[{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"ofa-sys/ofa","url":"https://github.com/ofa-sys/ofa"},{"title":"JHKim-snu/GVCCI","url":"https://github.com/JHKim-snu/GVCCI"},{"title":"JHKim-snu/PGA","url":"https://github.com/JHKim-snu/PGA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/prompt-tuning-for-generative-multimodal","title":"Prompt Tuning for Generative Multimodal Pretrained Models","date":"2022-08-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multimodal-adaptive-distillation-for","title":"Multimodal Adaptive Distillation for Leveraging Unimodal Encoders for Vision-Language Tasks","date":"2022-04-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simvlm-simple-visual-language-model","title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","date":"2021-08-24","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":18,"samples_unverified":19,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/how-much-can-clip-benefit-vision-and-language","title":"How Much Can CLIP Benefit Vision-and-Language Tasks?","date":"2021-07-13","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":6,"samples_unverified":4,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/seeing-out-of-the-box-end-to-end-pre-training","title":"Seeing Out of tHe bOx: End-to-End Pre-training for Vision-Language Representation Learning","date":"2021-04-07","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/large-scale-adversarial-training-for-vision","title":"Large-Scale Adversarial Training for Vision-and-Language Representation Learning","date":"2020-06-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":10,"samples_unverified":10,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visual-entailment-a-novel-task-for-fine","title":"Visual Entailment: A Novel Task for Fine-Grained Image Understanding","date":"2019-01-20","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":89,"samples_ran":48,"samples_unverified":41,"pointer_only_for_licence":46,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}