{"url":"/dataset/mmneedle","name":"MMNeedle","full_name":"Multimodal Needle in a Haystack","description_markdown":"We introduce the MultiModal Needle-in-a-haystack (MMNeedle) benchmark, specifically designed to assess the long-context capabilities of MLLMs. Besides multi-image input, we employ image stitching to further increase the input context length, and develop a protocol to automatically generate labels for sub-image level retrieval. Essentially, MMNeedle evaluates MLLMs by stress-testing their capability to locate a target sub-image (needle) within a set of images (haystack) based on textual instructions and descriptions of image contents. This setup necessitates an advanced understanding of extensive visual contexts and effective information retrieval within long-context image inputs.","description_withheld":null,"homepage":"https://github.com/Wang-ML-Lab/multimodal-needle-in-a-haystack/tree/main","introduced_date":"2024-06-17","introduced_date_note":null,"introduced_by":{"paper":"/paper/multimodal-needle-in-a-haystack-benchmarking","title":"Multimodal Needle in a Haystack: Benchmarking Long-Context Capability of Multimodal Large Language Models","first_author":"Hengyi Wang","url":null},"license":{"name":"CC BY 4.0","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Long-Context Understanding","url":"/task/long-context-understanding","datasets_with_task":"/datasets/task/long-context-understanding"},{"name":"multimodal generation","url":"/task/multimodal-generation","datasets_with_task":"/datasets/task/multimodal-generation"},{"name":"Hallucination","url":"/task/hallucination","datasets_with_task":"/datasets/task/hallucination"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["MMNeedle"],"data_loaders":[],"num_papers_in_archive":12,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/long-context-understanding-on-mmneedle","task":"Long-Context Understanding","dataset_variant":"MMNeedle","rows":12,"metrics":["1 Image, 4*4 Stitching, Exact Accuracy","1 Image, 8*8 Stitching, Exact Accuracy","1 Image, 2*2 Stitching, Exact Accuracy","10 Images, 1*1 Stitching, Exact Accuracy","10 Images, 2*2 Stitching, Exact Accuracy","10 Images, 4*4 Stitching, Exact Accuracy","10 Images, 8*8 Stitching, Exact Accuracy"],"first_row_in_archive_order":{"model":"GPT-4o","paper":"/paper/gpt-4-technical-report-1","metrics":{"1 Image, 2*2 Stitching, Exact Accuracy":"94.6","1 Image, 4*4 Stitching, Exact Accuracy":"83","1 Image, 8*8 Stitching, Exact Accuracy":"19","10 Images, 1*1 Stitching, Exact Accuracy":"97","10 Images, 2*2 Stitching, Exact Accuracy":"81.8","10 Images, 4*4 Stitching, Exact Accuracy":"26.9","10 Images, 8*8 Stitching, Exact Accuracy":"1"},"code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/what-matters-when-building-vision-language","title":"What matters when building vision-language models?","date":"2024-05-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/llava-uhd-an-lmm-perceiving-any-aspect-ratio","title":"LLaVA-UHD: an LMM Perceiving Any Aspect Ratio and High-Resolution Images","date":"2024-03-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/the-claude-3-model-family-opus-sonnet-haiku","title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","date":"2024-03-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mplug-owl2-revolutionizing-multi-modal-large","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","date":"2023-11-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cogvlm-visual-expert-for-pretrained-language","title":"CogVLM: Visual Expert for Pretrained Language Models","date":"2023-11-06","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/instructblip-towards-general-purpose-vision","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","date":"2023-05-11","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":4,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}