{"url":"/dataset/tvqa","name":"TVQA","full_name":"TVQA","description_markdown":"The **TVQA** dataset is a large-scale video dataset for video question answering. It is based on 6 popular TV shows (Friends, The Big Bang Theory, How I Met Your Mother, House M.D., Grey's Anatomy, Castle). It includes 152,545 QA pairs from 21,793 TV show clips. The QA pairs are split into the ratio of 8:1:1 for training, validation, and test sets. The TVQA dataset provides the sequence of video frames extracted at 3 FPS, the corresponding subtitles with the video clips, and the query consisting of a question and four answer candidates. Among the four answer candidates, there is only one correct answer.\r\n\r\nSource: [Two-stream Spatiotemporal Feature for Video QA Task](https://arxiv.org/abs/1907.05006)\r\nImage Source: [https://arxiv.org/abs/1809.01696](https://arxiv.org/abs/1809.01696)","description_withheld":null,"homepage":"http://tvqa.cs.unc.edu/","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/tvqa-localized-compositional-video-question","title":"TVQA: Localized, Compositional Video Question Answering","first_author":"Jie Lei","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Zero-Shot Video Question Answer","url":"/task/zeroshot-video-question-answer","datasets_with_task":"/datasets/task/zeroshot-video-question-answer"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["TVQA"],"data_loaders":[],"num_papers_in_archive":146,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-video-question-answer-on-tvqa","task":"Zero-Shot Video Question Answer","dataset_variant":"TVQA","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"FrozenBiLM (with speech)","paper":"/paper/zero-shot-video-question-answering-via-frozen","metrics":{"Accuracy":"59.7"},"code_links":[{"title":"antoyang/FrozenBiLM","url":"https://github.com/antoyang/FrozenBiLM"},{"title":"klauscc/dam","url":"https://github.com/klauscc/dam"},{"title":"sts-vlcc/sts-vlcc","url":"https://github.com/sts-vlcc/sts-vlcc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-question-answering-on-tvqa","task":"Video Question Answering","dataset_variant":"TVQA","rows":6,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"LLaMA-VQA","paper":"/paper/large-language-models-are-temporal-and-causal","metrics":{"Accuracy":"82.2"},"code_links":[{"title":"mlvlab/Flipped-VQA","url":"https://github.com/mlvlab/Flipped-VQA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-tvqa","task":"Zero-Shot Learning","dataset_variant":"TVQA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VideoChat2","paper":"/paper/mvbench-a-comprehensive-multi-modal-video","metrics":{"Accuracy":"40.6"},"code_links":[{"title":"opengvlab/ask-anything","url":"https://github.com/opengvlab/ask-anything"},{"title":"magic-research/PLLaVA","url":"https://github.com/magic-research/PLLaVA"},{"title":"bytedance/tarsier","url":"https://github.com/bytedance/tarsier"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/large-language-models-are-temporal-and-causal","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","date":"2023-10-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-chained-image-language-model-for-video-1","title":"Self-Chained Image-Language Model for Video Localization and Question Answering","date":"2023-05-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vindlu-a-recipe-for-effective-video-and","title":"VindLU: A Recipe for Effective Video-and-Language Pretraining","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":14,"samples_unverified":20,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/iperceive-applying-common-sense-reasoning-to-1","title":"iPerceive: Applying Common-Sense Reasoning to Multi-Modal Dense Video Captioning and Video Question Answering","date":"2020-11-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hero-hierarchical-encoder-for-video-language","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","date":"2020-05-01","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tvqa-spatio-temporal-grounding-for-video","title":"TVQA+: Spatio-Temporal Grounding for Video Question Answering","date":"2019-04-25","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":6,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":89,"samples_ran":46,"samples_unverified":43,"pointer_only_for_licence":16,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}