{"url":"/dataset/textvqa","name":"TextVQA","full_name":null,"description_markdown":"TextVQA is a dataset to benchmark visual reasoning based on text in images.\r\nTextVQA requires models to read and reason about text in images to answer questions about them. Specifically, models need to incorporate a new modality of text present in the images and reason over it to answer TextVQA questions.\r\n\r\nStatistics\r\n* 28,408 images from OpenImages\r\n* 45,336 questions\r\n* 453,360 ground truth answers","description_withheld":null,"homepage":"https://textvqa.org/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/towards-vqa-models-that-can-read","title":"Towards VQA Models That Can Read","first_author":"Amanpreet Singh","url":null},"license":{"name":"CC BY 4.0","url":"https://textvqa.org/dataset"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Visual Question Answering","url":"/task/visual-question-answering-1","datasets_with_task":"/datasets/task/visual-question-answering-1"}],"languages":[],"variants":["TextVQA Val","TextVQA Test","TextVQA test-standard","TextVQA"],"data_loaders":[],"num_papers_in_archive":476,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-textvqa-test-1","task":"Visual Question Answering (VQA)","dataset_variant":"TextVQA test-standard","rows":12,"metrics":["overall"],"first_row_in_archive_order":{"model":"PaLI","paper":"/paper/pali-a-jointly-scaled-multilingual-language","metrics":{"overall":"73.1"},"code_links":[{"title":"google-research/big_vision","url":"https://github.com/google-research/big_vision"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-textvqa-test-2","task":"Visual Question Answering","dataset_variant":"TextVQA test-standard","rows":1,"metrics":["overall"],"first_row_in_archive_order":{"model":"PromptCap","paper":"/paper/promptcap-prompt-guided-task-aware-image","metrics":{"overall":"51.80"},"code_links":[{"title":"Yushi-Hu/PromptCap","url":"https://github.com/Yushi-Hu/PromptCap"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-vqa-on-textvqa","task":"Visual Question Answering (VQA)","dataset_variant":"TextVQA","rows":1,"metrics":["Acc"],"first_row_in_archive_order":{"model":"Lyra-Pro","paper":"/paper/lyra-an-efficient-and-speech-centric","metrics":{"Acc":"83.5"},"code_links":[{"title":"dvlab-research/Lyra","url":"https://github.com/dvlab-research/Lyra"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":3,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/promptcap-prompt-guided-task-aware-image","title":"PromptCap: Prompt-Guided Task-Aware Image Captioning","date":"2022-11-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tag-boosting-text-vqa-via-text-aware-visual","title":"TAG: Boosting Text-VQA via Text-aware Visual Question-answer Generation","date":"2022-08-03","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":23,"samples_ran":5,"samples_unverified":18,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}