{"url":"/dataset/visual7w","name":"Visual7W","full_name":null,"description_markdown":"**Visual7W** is a large-scale visual question answering (QA) dataset, with object-level groundings and multimodal answers. Each question starts with one of the seven Ws, what, where, when, who, why, how and which. It is collected from 47,300 COCO images and it has 327,929 QA pairs, together with 1,311,756 human-generated multiple-choices and 561,459 object groundings from 36,579 categories.\r\n\r\nSource: [https://github.com/yukezhu/visual7w-toolkit](https://github.com/yukezhu/visual7w-toolkit)\r\nImage Source: [http://ai.stanford.edu/~yukez/visual7w/](http://ai.stanford.edu/~yukez/visual7w/)","description_withheld":null,"homepage":"http://ai.stanford.edu/~yukez/visual7w/","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/visual7w-grounded-question-answering-in","title":"Visual7W: Grounded Question Answering in Images","first_author":"Yuke Zhu","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Image Comprehension","url":"/task/image-comprehension","datasets_with_task":"/datasets/task/image-comprehension"}],"languages":[],"variants":["Visual7W"],"data_loaders":[],"num_papers_in_archive":112,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-visual7w","task":"Visual Question Answering (VQA)","dataset_variant":"Visual7W","rows":4,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"CMN","paper":"/paper/modeling-relationships-in-referential","metrics":{"Percentage correct":"72.53"},"code_links":[{"title":"hengyuan-hu/bottom-up-attention-vqa","url":"https://github.com/hengyuan-hu/bottom-up-attention-vqa"},{"title":"thilinicooray/Bottom-up-vqa","url":"https://github.com/thilinicooray/Bottom-up-vqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/coarse-to-fine-reasoning-for-visual-question","title":"Coarse-to-Fine Reasoning for Visual Question Answering","date":"2021-10-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/compact-trilinear-interaction-for-visual","title":"Compact Trilinear Interaction for Visual Question Answering","date":"2019-09-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/modeling-relationships-in-referential","title":"Modeling Relationships in Referential Expressions with Compositional Modular Networks","date":"2016-11-30","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/multimodal-compact-bilinear-pooling-for","title":"Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding","date":"2016-06-06","rows_on_this_dataset":1,"code_links":10,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}