{"url":"/dataset/infoseek","name":"InfoSeek","full_name":"Visual Information Seeking","description_markdown":"In this project, we introduce InfoSeek, a visual question answering dataset tailored for information-seeking questions that cannot be answered with only common sense knowledge. Using InfoSeek, we analyze various pre-trained visual question answering models and gain insights into their characteristics. Our findings reveal that state-of-the-art pre-trained multi-modal models (e.g., PaLI-X, BLIP2, etc.) face challenges in answering visual information-seeking questions, but fine-tuning on the InfoSeek dataset elicits models to use fine-grained knowledge that was learned during their pre-training.","description_withheld":null,"homepage":"https://open-vision-language.github.io/infoseek/","introduced_date":"2023-02-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/can-pre-trained-vision-and-language-models","title":"Can Pre-trained Vision and Language Models Answer Visual Information-Seeking Questions?","first_author":"Yang Chen","url":null},"license":{"name":"Apache-2.0","url":"https://github.com/open-vision-language/infoseek/blob/main/LICENSE"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Retrieval","url":"/task/retrieval","datasets_with_task":"/datasets/task/retrieval"},{"name":"Open-Domain Question Answering","url":"/task/open-domain-question-answering","datasets_with_task":"/datasets/task/open-domain-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["InfoSeek"],"data_loaders":[],"num_papers_in_archive":36,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-vqa-on-infoseek","task":"Visual Question Answering (VQA)","dataset_variant":"InfoSeek","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"RA-VQAv2 w/ PreFLMR","paper":"/paper/preflmr-scaling-up-fine-grained-late","metrics":{"Accuracy":"30.65"},"code_links":[{"title":"linweizhedragon/retrieval-augmented-visual-question-answering","url":"https://github.com/linweizhedragon/retrieval-augmented-visual-question-answering"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/retrieval-on-infoseek","task":"Retrieval","dataset_variant":"InfoSeek","rows":1,"metrics":["Recall@5"],"first_row_in_archive_order":{"model":"PreFLMR","paper":"/paper/preflmr-scaling-up-fine-grained-late","metrics":{"Recall@5":"62.1"},"code_links":[{"title":"linweizhedragon/retrieval-augmented-visual-question-answering","url":"https://github.com/linweizhedragon/retrieval-augmented-visual-question-answering"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/preflmr-scaling-up-fine-grained-late","title":"PreFLMR: Scaling Up Fine-Grained Late-Interaction Multi-modal Retrievers","date":"2024-02-13","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/can-pre-trained-vision-and-language-models","title":"Can Pre-trained Vision and Language Models Answer Visual Information-Seeking Questions?","date":"2023-02-23","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":3,"samples_harvested":21,"samples_ran":10,"samples_unverified":11,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}