{"url":"/dataset/how2qa","name":"How2QA","full_name":"How2QA","description_markdown":"To collect How2QA for video QA task, the same set of selected video clips are presented to another group of AMT workers for multichoice QA annotation. Each worker is assigned with one video segment and asked to write one question with four answer candidates (one correctand three distractors). Similarly, narrations are hidden from the workers to ensure the collected QA pairs are not biased by subtitles. Similar to TVQA, the start and end points are provided for the relevant moment for each question. After filtering low-quality annotations, the final dataset contains 44,007 QA pairs for 22k 60-second clips selected from 9035 videos.\r\n\r\nSource: [HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training](https://arxiv.org/pdf/2005.00200v2.pdf)","description_withheld":null,"homepage":"https://github.com/linjieli222/HERO","introduced_date":"2020-05-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/hero-hierarchical-encoder-for-video-language","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","first_author":"Linjie Li","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"}],"languages":[],"variants":["How2QA"],"data_loaders":[{"repo":"https://github.com/linjieli222/HERO","url":"https://github.com/linjieli222/HERO","frameworks":["pytorch"]}],"num_papers_in_archive":28,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-how2qa","task":"Video Question Answering","dataset_variant":"How2QA","rows":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Text + Text (no Multimodal Pretext Training)","paper":"/paper/towards-fast-adaptation-of-pretrained","metrics":{"Accuracy":"93.2"},"code_links":[{"title":"xudonglinthu/upgradable-multimodal-intelligence","url":"https://github.com/xudonglinthu/upgradable-multimodal-intelligence"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-how2qa","task":"Zero-Shot Learning","dataset_variant":"How2QA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"SeViLA","paper":null,"metrics":{"Accuracy":"72.3"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":14,"samples_unverified":20,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-fast-adaptation-of-pretrained","title":"Towards Fast Adaptation of Pretrained Contrastive Models for Multi-channel Video-Language Retrieval","date":"2022-06-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-the-video-in-video-language","title":"Revisiting the \"Video\" in Video-Language Understanding","date":"2022-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/just-ask-learning-to-answer-questions-from","title":"Just Ask: Learning to Answer Questions from Millions of Narrated Videos","date":"2020-12-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hero-hierarchical-encoder-for-video-language","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","date":"2020-05-01","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":53,"samples_ran":23,"samples_unverified":30,"pointer_only_for_licence":9,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}