{"url":"/dataset/nlvr","name":"NLVR","full_name":"Natural Language Visual Reasoningnatural language for visual reasoning","description_markdown":"**NLVR** contains 92,244 pairs of human-written English sentences grounded in synthetic images. Because the images are synthetically generated, this dataset can be used for semantic parsing.\r\n\r\nSource: [http://lil.nlp.cornell.edu/nlvr/](http://lil.nlp.cornell.edu/nlvr/)\r\nImage Source: [http://lil.nlp.cornell.edu/nlvr/](http://lil.nlp.cornell.edu/nlvr/)","description_withheld":null,"homepage":"http://lil.nlp.cornell.edu/nlvr/","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-corpus-of-natural-language-for-visual","title":"A Corpus of Natural Language for Visual Reasoning","first_author":"Alane Suhr","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Reasoning","url":"/task/visual-reasoning","datasets_with_task":"/datasets/task/visual-reasoning"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["NLVR","NLVR2 Dev","NLVR2 Test"],"data_loaders":[{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#nlvr","frameworks":["pytorch"]}],"num_papers_in_archive":83,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-reasoning-on-nlvr2-dev","task":"Visual Reasoning","dataset_variant":"NLVR2 Dev","rows":15,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"BEiT-3","paper":"/paper/image-as-a-foreign-language-beit-pretraining","metrics":{"Accuracy":"91.51"},"code_links":[{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm/tree/master/beit"},{"title":"lyan62/data-curation","url":"https://github.com/lyan62/data-curation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-reasoning-on-nlvr2-test","task":"Visual Reasoning","dataset_variant":"NLVR2 Test","rows":14,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"BEiT-3","paper":"/paper/image-as-a-foreign-language-beit-pretraining","metrics":{"Accuracy":"92.58"},"code_links":[{"title":"microsoft/unilm","url":"https://github.com/microsoft/unilm/tree/master/beit"},{"title":"lyan62/data-curation","url":"https://github.com/lyan62/data-curation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-reasoning-on-nlvr","task":"Visual Reasoning","dataset_variant":"NLVR","rows":1,"metrics":["Accuracy (Dev)","Accuracy (Test-P)","Accuracy (Test-U)"],"first_row_in_archive_order":{"model":"VisualBERT","paper":"/paper/visualbert-a-simple-and-performant-baseline","metrics":{"Accuracy (Dev)":"67.4%","Accuracy (Test-P)":"67%","Accuracy (Test-U)":"67.3%"},"code_links":[{"title":"uclanlp/visualbert","url":"https://github.com/uclanlp/visualbert"},{"title":"YIKUAN8/Transformers-VQA","url":"https://github.com/YIKUAN8/Transformers-VQA"},{"title":"lalithjets/surgical_vqa","url":"https://github.com/lalithjets/surgical_vqa"},{"title":"gchhablani/multilingual-vqa","url":"https://github.com/gchhablani/multilingual-vqa"},{"title":"longbai1006/surgical-vqla","url":"https://github.com/longbai1006/surgical-vqla"},{"title":"social-ai-studio/matk","url":"https://github.com/social-ai-studio/matk"},{"title":"chenkangyang/paddle_visual_bert","url":"https://github.com/chenkangyang/paddle_visual_bert"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/5/visual_bert"},{"title":"MindCode-4/code-5","url":"https://github.com/MindCode-4/code-5/tree/main/visual_bert"},{"title":"MindCode-4/code-1","url":"https://github.com/MindCode-4/code-1/tree/main/visual_bert"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/differentiable-outlier-detection-enable","title":"Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-02-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/toward-building-general-foundation-models-for","title":"Toward Building General Foundation Models for Language, Vision, and Vision-Language Understanding Tasks","date":"2023-01-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/image-as-a-foreign-language-beit-pretraining","title":"Image as a Foreign Language: BEiT Pretraining for All Vision and Vision-Language Tasks","date":"2022-08-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/blip-bootstrapping-language-image-pre","title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation","date":"2022-01-28","rows_on_this_dataset":1,"code_links":9,"syntology":null},{"paper":"/paper/multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vlmo-unified-vision-language-pre-training","title":"VLMo: Unified Vision-Language Pre-Training with Mixture-of-Modality-Experts","date":"2021-11-03","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/simvlm-simple-visual-language-model","title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","date":"2021-08-24","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":18,"samples_unverified":19,"pointer_only_for_licence":28,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/seeing-out-of-the-box-end-to-end-pre-training","title":"Seeing Out of tHe bOx: End-to-End Pre-training for Vision-Language Representation Learning","date":"2021-04-07","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lxmert-learning-cross-modality-encoder","title":"LXMERT: Learning Cross-Modality Encoder Representations from Transformers","date":"2019-08-20","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":4,"samples_unverified":11,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visualbert-a-simple-and-performant-baseline","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","date":"2019-08-09","rows_on_this_dataset":2,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":97,"samples_ran":45,"samples_unverified":52,"pointer_only_for_licence":49,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}