{"url":"/dataset/marvl","name":"MaRVL","full_name":"Multicultural Reasoning over Vision and Language","description_markdown":"**M**ulticultural **R**easoning over **V**ision and **L**anguage (MaRVL) is a dataset based on an ImageNet-style hierarchy representative of many languages and cultures (Indonesian, Mandarin Chinese, Swahili, Tamil, and Turkish). The selection of both concepts and images is entirely driven by native speakers. Afterwards, we elicit statements from native speakers about pairs of images. The task consists in discriminating whether each grounded statement is true or false.","description_withheld":null,"homepage":"https://marvl-challenge.github.io/","introduced_date":"2021-09-28","introduced_date_note":null,"introduced_by":{"paper":"/paper/visually-grounded-reasoning-across-languages","title":"Visually Grounded Reasoning across Languages and Cultures","first_author":"Fangyu Liu","url":null},"license":{"name":"CC BY 4.0 license","url":"https://marvl-challenge.github.io/download#terms-of-use"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Reasoning","url":"/task/visual-reasoning","datasets_with_task":"/datasets/task/visual-reasoning"},{"name":"Zero-Shot Cross-Lingual Transfer","url":"/task/zero-shot-cross-lingual-transfer","datasets_with_task":"/datasets/task/zero-shot-cross-lingual-transfer"},{"name":"Zero-Shot Cross-Lingual Visual Reasoning","url":"/task/zero-shot-cross-lingual-visual-reasoning","datasets_with_task":"/datasets/task/zero-shot-cross-lingual-visual-reasoning"},{"name":"Max-Shot Cross-Lingual Visual Reasoning","url":"/task/max-shot-cross-lingual-visual-reasoning","datasets_with_task":"/datasets/task/max-shot-cross-lingual-visual-reasoning"}],"languages":[{"name":"Chinese","url":"/datasets/language/chinese"},{"name":"Indonesian","url":"/datasets/language/indonesian"},{"name":"Tamil","url":"/datasets/language/tamil"},{"name":"Turkish","url":"/datasets/language/turkish"},{"name":"Swahili","url":"/datasets/language/swahili"}],"variants":["MaRVL"],"data_loaders":[],"num_papers_in_archive":29,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-cross-lingual-transfer-on-marvl","task":"Zero-Shot Cross-Lingual Transfer","dataset_variant":"MaRVL","rows":2,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"xUNITER","paper":"/paper/visually-grounded-reasoning-across-languages","metrics":{"Accuracy (%)":"56.1"},"code_links":[{"title":"e-bug/volta","url":"https://github.com/e-bug/volta"},{"title":"marvl-challenge/marvl-code","url":"https://github.com/marvl-challenge/marvl-code"},{"title":"shin-ee-chen/bla","url":"https://github.com/shin-ee-chen/bla"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/visually-grounded-reasoning-across-languages","title":"Visually Grounded Reasoning across Languages and Cultures","date":"2021-09-28","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}