{"url":"/dataset/musicqa-dataset","name":"MMVP","full_name":null,"description_markdown":"The **MMVP (Multimodal Visual Patterns) Benchmark** focuses on identifying \"CLIP-blind pairs\" – images that appear similar to the CLIP model despite having clear visual differences. These patterns highlight the challenges these systems face in answering straightforward questions, often leading to incorrect responses and hallucinated explanations.","description_withheld":null,"homepage":"https://tsb0601.github.io/mmvp_blog","introduced_date":"2024-01-11","introduced_date_note":null,"introduced_by":{"paper":"/paper/eyes-wide-shut-exploring-the-visual","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","first_author":"Shengbang Tong","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Spatial Reasoning","url":"/task/spatial-reasoning","datasets_with_task":"/datasets/task/spatial-reasoning"},{"name":"Chart Question Answering","url":"/task/chart-question-answering","datasets_with_task":"/datasets/task/chart-question-answering"},{"name":"Music Question Answering","url":"/task/music-question-answering","datasets_with_task":"/datasets/task/music-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MMVP"],"data_loaders":[],"num_papers_in_archive":53,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/spatial-reasoning-on-mmvp","task":"Spatial Reasoning","dataset_variant":"MMVP","rows":0,"metrics":["Overall Success Rate"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}