{"url":"/dataset/infographicvqa","name":"InfographicVQA","full_name":null,"description_markdown":"**InfographicVQA** is a dataset that comprises a diverse collection of infographics along with natural language questions and answers annotations. The collected questions require methods to jointly reason over the document layout, textual content, graphical elements, and data visualizations. We curate the dataset with emphasis on questions that require elementary reasoning and basic arithmetic skills.","description_withheld":null,"homepage":"https://www.docvqa.org/datasets/infographicvqa","introduced_date":"2021-04-26","introduced_date_note":null,"introduced_by":{"paper":"/paper/infographicvqa","title":"InfographicVQA","first_author":"Minesh Mathew","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["InfographicVQA"],"data_loaders":[],"num_papers_in_archive":52,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-vqa-on","task":"Visual Question Answering (VQA)","dataset_variant":"InfographicVQA","rows":21,"metrics":["ANLS"],"first_row_in_archive_order":{"model":"Gemini Ultra (pixel only)","paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","metrics":{"ANLS":"80.3"},"code_links":[{"title":"valdecy/pybibx","url":"https://github.com/valdecy/pybibx"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/lapdoc-layout-aware-prompting-for-documents","title":"LAPDoc: Layout-Aware Prompting for Documents","date":"2024-02-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/screenai-a-vision-language-model-for-ui-and","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","date":"2024-02-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/omni-smola-boosting-generalist-multimodal","title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","date":"2023-12-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/pali-3-vision-language-models-smaller-faster","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","date":"2023-10-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/docformerv2-local-features-for-document","title":"DocFormerv2: Local Features for Document Understanding","date":"2023-06-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/layout-and-task-aware-instruction-prompt-for","title":"Layout and Task Aware Instruction Prompt for Zero-shot Document Image Question Answering","date":"2023-06-01","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dublin-document-understanding-by-language","title":"DUBLIN -- Document Understanding By Language-Image Network","date":"2023-05-23","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/matcha-enhancing-visual-language-pretraining","title":"MatCha: Enhancing Visual Language Pretraining with Math Reasoning and Chart Derendering","date":"2022-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unifying-vision-text-and-layout-for-universal","title":"Unifying Vision, Text, and Layout for Universal Document Processing","date":"2022-12-05","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":4,"samples_unverified":13,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":5,"samples_harvested":42,"samples_ran":14,"samples_unverified":28,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}