{"url":"/dataset/ivqa","name":"iVQA","full_name":"Instructional Video Question Answering","description_markdown":"An open-ended VideoQA benchmark that aims to: i) provide a well-defined evaluation by including five correct answer annotations per question and ii) avoid questions which can be answered without the video. \r\n\r\niVQA contains 10,000 video clips with one question and five corresponding answers per clip. Moreover, we manually reduce the language bias by excluding questions that could be answered without watching the video.  \r\n\r\nSource: [Just Ask: Learning to Answer Questions from Millions of Narrated Videos](https://antoyang.github.io/just-ask.html)","description_withheld":null,"homepage":"https://antoyang.github.io/just-ask.html","introduced_date":"2020-12-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/just-ask-learning-to-answer-questions-from","title":"Just Ask: Learning to Answer Questions from Millions of Narrated Videos","first_author":"Antoine Yang","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["iVQA"],"data_loaders":[{"repo":"https://github.com/antoyang/just-ask","url":"https://github.com/antoyang/just-ask","frameworks":[]}],"num_papers_in_archive":22,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-ivqa","task":"Video Question Answering","dataset_variant":"iVQA","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Text + Text (no Multimodal Pretext Training)","paper":"/paper/towards-fast-adaptation-of-pretrained","metrics":{"Accuracy":"40.2"},"code_links":[{"title":"xudonglinthu/upgradable-multimodal-intelligence","url":"https://github.com/xudonglinthu/upgradable-multimodal-intelligence"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-ivqa","task":"Zero-Shot Learning","dataset_variant":"iVQA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"FrozenBiLM","paper":"/paper/zero-shot-video-question-answering-via-frozen","metrics":{"Accuracy":"0.268"},"code_links":[{"title":"antoyang/FrozenBiLM","url":"https://github.com/antoyang/FrozenBiLM"},{"title":"klauscc/dam","url":"https://github.com/klauscc/dam"},{"title":"sts-vlcc/sts-vlcc","url":"https://github.com/sts-vlcc/sts-vlcc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/video-question-answering-with-iterative-video","title":"Video Question Answering with Iterative Video-Text Co-Tokenization","date":"2022-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":14,"samples_unverified":20,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-fast-adaptation-of-pretrained","title":"Towards Fast Adaptation of Pretrained Contrastive Models for Multi-channel Video-Language Retrieval","date":"2022-06-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/just-ask-learning-to-answer-questions-from","title":"Just Ask: Learning to Answer Questions from Millions of Narrated Videos","date":"2020-12-01","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":35,"samples_ran":15,"samples_unverified":20,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}