{"url":"/dataset/docvqa","name":"DocVQA","full_name":null,"description_markdown":"DocVQA consists of 50,000 questions defined on 12,000+ document images.\r\n\r\nSource: [DocVQA: A Dataset for VQA on Document Images](/paper/docvqa-a-dataset-for-vqa-on-document-images)","description_withheld":null,"homepage":"https://cvit.iiit.ac.in/docvqa/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/docvqa-a-dataset-for-vqa-on-document-images","title":"DocVQA: A Dataset for VQA on Document Images","first_author":"Minesh Mathew","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Reading Comprehension","url":"/task/reading-comprehension","datasets_with_task":"/datasets/task/reading-comprehension"}],"languages":[],"variants":["DocVQA val","DocVQA test","DocVQA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/eliolio/docvqa","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":290,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-docvqa-test","task":"Visual Question Answering (VQA)","dataset_variant":"DocVQA test","rows":33,"metrics":["ANLS","Accuracy"],"first_row_in_archive_order":{"model":"Human","paper":"/paper/docvqa-a-dataset-for-vqa-on-document-images","metrics":{"ANLS":"0.9436"},"code_links":[{"title":"anisha2102/docvqa","url":"https://github.com/anisha2102/docvqa"},{"title":"mineshmathew/DocVQA","url":"https://github.com/mineshmathew/DocVQA/tree/master/BERT_baseline"},{"title":"mineshmathew/DocVQA","url":"https://github.com/mineshmathew/DocVQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-docvqa-val","task":"Visual Question Answering (VQA)","dataset_variant":"DocVQA val","rows":2,"metrics":["Accuracy","bk lôn"],"first_row_in_archive_order":{"model":"BERT LARGE Baseline","paper":"/paper/docvqa-a-dataset-for-vqa-on-document-images","metrics":{"Accuracy":"54.48"},"code_links":[{"title":"anisha2102/docvqa","url":"https://github.com/anisha2102/docvqa"},{"title":"mineshmathew/DocVQA","url":"https://github.com/mineshmathew/DocVQA/tree/master/BERT_baseline"},{"title":"mineshmathew/DocVQA","url":"https://github.com/mineshmathew/DocVQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-vqa-on-docvqa","task":"Visual Question Answering (VQA)","dataset_variant":"DocVQA","rows":1,"metrics":["ANLS"],"first_row_in_archive_order":{"model":"ChatGPT 3.5 with LAPDoc Prompt (SpatialFormat)","paper":"/paper/lapdoc-layout-aware-prompting-for-documents","metrics":{"ANLS":"79.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/multi-label-cluster-discrimination-for-visual","title":"Multi-label Cluster Discrimination for Visual Representation Learning","date":"2024-07-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lapdoc-layout-aware-prompting-for-documents","title":"LAPDoc: Layout-Aware Prompting for Documents","date":"2024-02-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/screenai-a-vision-language-model-for-ui-and","title":"ScreenAI: A Vision-Language Model for UI and Infographics Understanding","date":"2024-02-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/omni-smola-boosting-generalist-multimodal","title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","date":"2023-12-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/pali-3-vision-language-models-smaller-faster","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","date":"2023-10-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/docformerv2-local-features-for-document","title":"DocFormerv2: Local Features for Document Understanding","date":"2023-06-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/layout-and-task-aware-instruction-prompt-for","title":"Layout and Task Aware Instruction Prompt for Zero-shot Document Image Question Answering","date":"2023-06-01","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dublin-document-understanding-by-language","title":"DUBLIN -- Document Understanding By Language-Image Network","date":"2023-05-23","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/matcha-enhancing-visual-language-pretraining","title":"MatCha: Enhancing Visual Language Pretraining with Math Reasoning and Chart Derendering","date":"2022-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/unifying-vision-text-and-layout-for-universal","title":"Unifying Vision, Text, and Layout for Universal Document Processing","date":"2022-12-05","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":4,"samples_unverified":13,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ernie-layout-layout-knowledge-enhanced-pre","title":"ERNIE-Layout: Layout Knowledge Enhanced Pre-training for Visually-rich Document Understanding","date":"2022-10-12","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-document-recognition-and","title":"End-to-end Document Recognition and Understanding with Dessurt","date":"2022-03-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","rows_on_this_dataset":2,"code_links":9,"syntology":null},{"paper":"/paper/docvqa-a-dataset-for-vqa-on-document-images","title":"DocVQA: A Dataset for VQA on Document Images","date":"2020-07-01","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":11,"samples_harvested":88,"samples_ran":25,"samples_unverified":63,"pointer_only_for_licence":5,"papers_with_no_sample_that_ran":4,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}