{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/docvqa-a-dataset-for-vqa-on-document-images","title":"DocVQA: A Dataset for VQA on Document Images","arxiv_id":"2007.00398","date":"2020-07-01","proceeding":null,"authors":["Minesh Mathew","Dimosthenis Karatzas","C. V. Jawahar"],"abstract":"We present a new dataset for Visual Question Answering (VQA) on document images called DocVQA. The dataset consists of 50,000 questions defined on 12,000+ document images. Detailed analysis of the dataset in comparison with similar datasets for VQA and reading comprehension is presented. We report several baseline results by adopting existing VQA and reading comprehension models. Although the existing models perform reasonably well on certain types of questions, there is large performance gap compared to human performance (94.36% accuracy). The models need to improve specifically on questions where understanding structure of the document is crucial. The dataset, code and leaderboard are available at docvqa.org","url_abs":"https://arxiv.org/abs/2007.00398v3","url_pdf":"https://arxiv.org/pdf/2007.00398v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"docvqa-a-dataset-for-vqa-on-document-images","repo_url":"https://github.com/mineshmathew/DocVQA","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"docvqa-a-dataset-for-vqa-on-document-images","repo_url":"https://github.com/anisha2102/docvqa","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"docvqa-a-dataset-for-vqa-on-document-images","repo_url":"https://github.com/mineshmathew/DocVQA/tree/master/BERT_baseline","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"question-answering","task_name":"Question Answering"},{"task_slug":"reading-comprehension","task_name":"Reading Comprehension"},{"task_slug":"visual-question-answering-1","task_name":"Visual Question Answering"},{"task_slug":"visual-question-answering","task_name":"Visual Question Answering (VQA)"}],"methods":[{"method_slug":"bert","method_name":"BERT"}],"datasets_introduced":[{"slug":"docvqa","name":"DocVQA","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/visual-question-answering-on-docvqa-test","task":"Visual Question Answering (VQA)","dataset":"DocVQA test","model":"Human","rank_in_archive_order":1,"of":33,"metrics":{"ANLS":"0.9436"},"uses_additional_data":true},{"leaderboard":"/sota/visual-question-answering-on-docvqa-test","task":"Visual Question Answering (VQA)","dataset":"DocVQA test","model":"BERT_LARGE_SQUAD_DOCVQA_FINETUNED_Baseline","rank_in_archive_order":30,"of":33,"metrics":{"ANLS":"0.665","Accuracy":"55.77"},"uses_additional_data":true},{"leaderboard":"/sota/visual-question-answering-on-docvqa-val","task":"Visual Question Answering (VQA)","dataset":"DocVQA val","model":"BERT LARGE Baseline","rank_in_archive_order":1,"of":2,"metrics":{"Accuracy":"54.48"},"uses_additional_data":false},{"leaderboard":"/sota/visual-question-answering-on-docvqa-val","task":"Visual Question Answering (VQA)","dataset":"DocVQA val","model":"đm bk","rank_in_archive_order":2,"of":2,"metrics":{"bk lôn":"0.655"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2007.00398","atlas_url":"https://app.syntology.ai/?focus=2007.00398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.00398"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/anisha2102/docvqa","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mineshmathew/DocVQA/tree/master/BERT_baseline","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mineshmathew/DocVQA","reach":{"status":"ok"}}],"summary":{"ran_draft_wrong":2,"ran":2,"unverified":2},"by_repo_kind":{"listed":{"samples":6,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"9df40357afea56cc","entry":"to_list","repo":"anisha2102/docvqa","repo_kind":"listed","path":"run_docvqa.py","file_url":"https://github.com/anisha2102/docvqa/blob/HEAD/run_docvqa.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":2,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"9df40357afea56cc"}},{"code_sha256_prefix":"33893096d0c88812","entry":"bbox_string","repo":"anisha2102/docvqa","repo_kind":"listed","path":"create_dataset.py","file_url":"https://github.com/anisha2102/docvqa/blob/HEAD/create_dataset.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"33893096d0c88812"}},{"code_sha256_prefix":"58efca4ff69d4f1c","entry":"clean_text","repo":"anisha2102/docvqa","repo_kind":"listed","path":"create_dataset.py","file_url":"https://github.com/anisha2102/docvqa/blob/HEAD/create_dataset.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"58efca4ff69d4f1c"}},{"code_sha256_prefix":"1923fc05163d207d","entry":"convert_to_unicode","repo":"anisha2102/docvqa","repo_kind":"listed","path":"tokenization.py","file_url":"https://github.com/anisha2102/docvqa/blob/HEAD/tokenization.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"1923fc05163d207d"}},{"code_sha256_prefix":"ff83ccc8b0b6462d","entry":"load_vocab","repo":"anisha2102/docvqa","repo_kind":"listed","path":"tokenization.py","file_url":"https://github.com/anisha2102/docvqa/blob/HEAD/tokenization.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ff83ccc8b0b6462d"}},{"code_sha256_prefix":"0e5615f8994003cf","entry":"printable_text","repo":"anisha2102/docvqa","repo_kind":"listed","path":"tokenization.py","file_url":"https://github.com/anisha2102/docvqa/blob/HEAD/tokenization.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0e5615f8994003cf"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}