{"url":"/dataset/vizwiz","name":"VizWiz","full_name":"VizWiz-VQA","description_markdown":"The **VizWiz**-VQA dataset originates from a natural visual question answering setting where blind people each took an image and recorded a spoken question about it, together with 10 crowdsourced answers per visual question. The proposed challenge addresses the following two tasks for this dataset: predict the answer to a visual question and (2) predict whether a visual question cannot be answered.\r\n\r\nSource: [https://vizwiz.org/tasks-and-datasets/vqa/](https://vizwiz.org/tasks-and-datasets/vqa/)\r\nImage Source: [https://vizwiz.org/tasks-and-datasets/vqa/](https://vizwiz.org/tasks-and-datasets/vqa/)","description_withheld":null,"homepage":"https://vizwiz.org/tasks-and-datasets/vqa/","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/vizwiz-grand-challenge-answering-visual","title":"VizWiz Grand Challenge: Answering Visual Questions from Blind People","first_author":"Danna Gurari","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Visual Question Answering","url":"/task/visual-question-answering-1","datasets_with_task":"/datasets/task/visual-question-answering-1"},{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"}],"languages":[],"variants":["VizWiz Answer Differences 2019","VizWiz 2020 VQA","VizWiz 2020 test-dev","VizWiz 2020 test","VizWiz 2020 Answerability","VizWiz 2018 Answerability","VizWiz 2018","VizWiz"],"data_loaders":[],"num_papers_in_archive":260,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/image-captioning-on-vizwiz-2020-test-dev","task":"Image Captioning","dataset_variant":"VizWiz 2020 test-dev","rows":50,"metrics":["CIDEr","B1","B2","B3","B4","ROUGE-L","METEOR","SPICE"],"first_row_in_archive_order":{"model":"IBM Research AI","paper":null,"metrics":{"B1":"72.48","B2":"53.93","B3":"38.79","B4":"27.35","CIDEr":"80.67","METEOR":"22.21","ROUGE-L":"50.11","SPICE":"16.96"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vizwiz-2020-vqa","task":"Visual Question Answering (VQA)","dataset_variant":"VizWiz 2020 VQA","rows":16,"metrics":["overall","yes/no","number","other","unanswerable"],"first_row_in_archive_order":{"model":"PaLI","paper":"/paper/pali-a-jointly-scaled-multilingual-language","metrics":{"overall":"73.3"},"code_links":[{"title":"google-research/big_vision","url":"https://github.com/google-research/big_vision"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-vizwiz-2020-test","task":"Image Captioning","dataset_variant":"VizWiz 2020 test","rows":13,"metrics":["CIDEr","B1","B2","B3","B4","ROUGE-L","METEOR","SPICE"],"first_row_in_archive_order":{"model":"IBM Research AI","paper":null,"metrics":{"B1":"72.77","B2":"54.17","B3":"38.97","B4":"27.44","CIDEr":"81.04","METEOR":"22.25","ROUGE-L":"50.2","SPICE":"17.0"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vizwiz-2018-1","task":"Visual Question Answering (VQA)","dataset_variant":"VizWiz 2018","rows":10,"metrics":["overall","yes/no","number","other","unanswerable"],"first_row_in_archive_order":{"model":"LXR955, No Ensemble","paper":"/paper/lxmert-learning-cross-modality-encoder","metrics":{"number":"24.76","other":"39.0","overall":"55.4","unanswerable":"82.26","yes/no":"74.0"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"airsplay/lxmert","url":"https://github.com/airsplay/lxmert"},{"title":"zhegan27/VILLA","url":"https://github.com/zhegan27/VILLA"},{"title":"zhegan27/LXMERT-AdvTrain","url":"https://github.com/zhegan27/LXMERT-AdvTrain"},{"title":"social-ai-studio/matk","url":"https://github.com/social-ai-studio/matk"},{"title":"itsShnik/adaptively-finetuning-transformers","url":"https://github.com/itsShnik/adaptively-finetuning-transformers"},{"title":"ghazaleh-mahmoodi/lxmert_compression","url":"https://github.com/ghazaleh-mahmoodi/lxmert_compression"},{"title":"chaitanyadwivedii/3D-Attention-is-All-You-Need","url":"https://github.com/chaitanyadwivedii/3D-Attention-is-All-You-Need"},{"title":"Mind23-2/MindCode-156","url":"https://github.com/Mind23-2/MindCode-156"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vizwiz-2020","task":"Visual Question Answering (VQA)","dataset_variant":"VizWiz 2020 Answerability","rows":6,"metrics":["average_precision","f1_score"],"first_row_in_archive_order":{"model":"CLIP-Ensemble","paper":"/paper/less-is-more-linear-layers-on-clip-features","metrics":{"average_precision":"84.13"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vizwiz-1","task":"Visual Question Answering","dataset_variant":"VizWiz","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Emu-I *","paper":"/paper/generative-pretraining-in-multimodality","metrics":{"Accuracy":"38.1"},"code_links":[{"title":"baaivision/emu","url":"https://github.com/baaivision/emu"},{"title":"doc-doc/NExT-OE","url":"https://github.com/doc-doc/NExT-OE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-vizwiz-2018","task":"Visual Question Answering (VQA)","dataset_variant":"VizWiz 2018 Answerability","rows":1,"metrics":["average_precision","f1_score"],"first_row_in_archive_order":{"model":"ensemble_two_best","paper":null,"metrics":{"average_precision":"82.78","f1_score":"68.72"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/generative-pretraining-in-multimodality","title":"Emu: Generative Pretraining in Multimodality","date":"2023-07-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/less-is-more-linear-layers-on-clip-features","title":"Less Is More: Linear Layers on CLIP Features as Powerful VizWiz Model","date":"2022-06-10","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/decoupled-box-proposal-and-featurization-with","title":"Decoupled Box Proposal and Featurization with Ultrafine-Grained Semantic Labels Improve Image Captioning and Visual Question Answering","date":"2019-09-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/lxmert-learning-cross-modality-encoder","title":"LXMERT: Learning Cross-Modality Encoder Representations from Transformers","date":"2019-08-20","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":4,"samples_unverified":11,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-vqa-models-that-can-read","title":"Towards VQA Models That Can Read","date":"2019-04-18","rows_on_this_dataset":1,"code_links":7,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":26,"samples_ran":9,"samples_unverified":17,"pointer_only_for_licence":8,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}