{"url":"/dataset/valse","name":"VALSE","full_name":"VALSE: A Task-Independent Benchmark for Vision and Language Models Centered on Linguistic Phenomena","description_markdown":"We propose VALSE (Vision And Language Structured Evaluation), a novel benchmark designed for testing general-purpose pretrained vision and language (V&L) models for their visio-linguistic grounding capabilities on specific linguistic phenomena. VALSE offers a suite of six tests covering various linguistic constructs. Solving these requires models to ground linguistic phenomena in the visual modality, allowing more fine-grained evaluations than hitherto possible.\r\nWe expect VALSE to serve as an important benchmark to measure future progress of pretrained V&L models from a linguistic perspective, complementing the canonical task-centred V&L evaluations.","description_withheld":null,"homepage":"https://github.com/Heidelberg-NLP/VALSE","introduced_date":"2021-12-14","introduced_date_note":null,"introduced_by":{"paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","title":"VALSE: A Task-Independent Benchmark for Vision and Language Models Centered on Linguistic Phenomena","first_author":"Letitia Parcalabescu","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"image-sentence alignment","url":"/task/image-sentence-alignment","datasets_with_task":"/datasets/task/image-sentence-alignment"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["VALSE foil-it (noun phrases)","VALSE coreference clean","VALSE coreference standard","VALSE actant swap","VALSE action replacement","VALSE spatial relations","VALSE counting adversarial","VALSE counting small numbers","VALSE counting balanced","VALSE plurality","VALSE existence","VALSE"],"data_loaders":[],"num_papers_in_archive":26,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/image-sentence-alignment-on-valse","task":"image-sentence alignment","dataset_variant":"VALSE","rows":7,"metrics":["average pairwise accuracy","Average Accuracy"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Average Accuracy":"63.2","average pairwise accuracy":"75.1"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-actant-swap","task":"image-sentence alignment","dataset_variant":"VALSE actant swap","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"GPT2","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"pairwise accuracy":"76.9"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-action","task":"image-sentence alignment","dataset_variant":"VALSE action replacement","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"CLIP","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"pairwise accuracy":"75.6"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-coreference","task":"image-sentence alignment","dataset_variant":"VALSE coreference standard","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Accuracy (%)":"54.4","pairwise accuracy":"75.7"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-coreference-1","task":"image-sentence alignment","dataset_variant":"VALSE coreference clean","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Accuracy (%)":"54.3","pairwise accuracy":"69.2"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-counting","task":"image-sentence alignment","dataset_variant":"VALSE counting balanced","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Accuracy (%)":"64.9","pairwise accuracy":"76.7"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-counting-1","task":"image-sentence alignment","dataset_variant":"VALSE counting small numbers","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Accuracy (%)":"69.2","pairwise accuracy":"80.2"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-counting-2","task":"image-sentence alignment","dataset_variant":"VALSE counting adversarial","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Accuracy (%)":"66.7","pairwise accuracy":"77.3"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-existence","task":"image-sentence alignment","dataset_variant":"VALSE existence","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Accuracy (%)":"89.0","pairwise accuracy":"95.6"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-foil-it","task":"image-sentence alignment","dataset_variant":"VALSE foil-it (noun phrases)","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"CLIP","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"pairwise accuracy":"88.8"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-plurality","task":"image-sentence alignment","dataset_variant":"VALSE plurality","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"ViLBERT 12-in-1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"Accuracy (%)":"62.0","pairwise accuracy":"72.4"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-sentence-alignment-on-valse-spatial","task":"image-sentence alignment","dataset_variant":"VALSE spatial relations","rows":7,"metrics":["pairwise accuracy","Accuracy (%)"],"first_row_in_archive_order":{"model":"GPT1","paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","metrics":{"pairwise accuracy":"77.2"},"code_links":[{"title":"heidelberg-nlp/valse","url":"https://github.com/heidelberg-nlp/valse"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/valse-a-task-independent-benchmark-for-vision-1","title":"VALSE: A Task-Independent Benchmark for Vision and Language Models Centered on Linguistic Phenomena","date":"2021-12-14","rows_on_this_dataset":84,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}