{"url":"/dataset/vqg","name":"VQG","full_name":"Visual Question Generation","description_markdown":"**VQG** is a collection of datasets for visual question generation. VQG questions were collected by crowdsourcing the task on Amazon Mechanical Turk (AMT). The authors provided details on the prompt and the specific instructions for all the crowdsourcing tasks in this paper in the supplementary material. The prompt was successful at capturing nonliteral questions. Images were taken from the MSCOCO dataset.\r\n\r\nSource: [What BERT Sees: Cross-Modal Transfer for Visual Question Generation](https://arxiv.org/pdf/2002.10832v3.pdf)","description_withheld":null,"homepage":"https://www.microsoft.com/en-us/download/details.aspx?id=53670","introduced_date":"2016-03-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/generating-natural-questions-about-an-image","title":"Generating Natural Questions About an Image","first_author":"Nasrin Mostafazadeh","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Question Generation","url":"/task/question-generation","datasets_with_task":"/datasets/task/question-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Visual Question Generation","VQG"],"data_loaders":[],"num_papers_in_archive":80,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-generation-on-visual-question","task":"Question Generation","dataset_variant":"Visual Question Generation","rows":1,"metrics":["BLEU-1"],"first_row_in_archive_order":{"model":"MDN","paper":"/paper/multimodal-differential-network-for-visual","metrics":{"BLEU-1":"36.0"},"code_links":[{"title":"badripatro/MDN-VQG","url":"https://github.com/badripatro/MDN-VQG"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/multimodal-differential-network-for-visual","title":"Multimodal Differential Network for Visual Question Generation","date":"2018-08-12","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}