{"url":"/dataset/visual-haystacks-vhs","name":"Visual Haystacks (VHs)","full_name":null,"description_markdown":"Visual Haystacks (VHs) is a \"visual-centric\" Needle-In-A-Haystack (NIAH) benchmark specifically designed to evaluate the capabilities of Large Multimodal Models (LMMs) in visual retrieval and reasoning over sets of unrelated images. Unlike conventional NIAH challenges that center on text-related retrieval and understanding with limited anecdotal examples, VHs contains a much larger number of examples and focuses on \"simple visual tasks\", providing a more accurate reflection of LMMs' capabilities when dealing with extensive visual context.","description_withheld":null,"homepage":"https://visual-haystacks.github.io/","introduced_date":"2024-07-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/visual-haystacks-answering-harder-questions","title":"Visual Haystacks: A Vision-Centric Needle-In-A-Haystack Benchmark","first_author":"Tsung-Han Wu","url":null},"license":{"name":"MIT","url":"https://github.com/visual-haystacks/vhs_benchmark/blob/main/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Retrieval","url":"/task/retrieval","datasets_with_task":"/datasets/task/retrieval"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Visual Haystacks (VHs)"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}