{"url":"/dataset/msvd-qa","name":"MSVD-QA","full_name":null,"description_markdown":"The MSVD-QA dataset is a Video Question Answering (VideoQA) dataset. It is based on the existing Microsoft Research Video Description (MSVD) dataset, which consists of about 120K sentences describing more than 2,000 video snippets. In the MSVD-QA dataset, Question-Answer (QA) pairs are generated from these descriptions. The dataset is mainly used in video captioning experiments but due to its large data size, it is also used for VideoQA. It contains 1970 video clips and approximately 50.5K QA pairs.","description_withheld":null,"homepage":"https://github.com/xudejing/video-question-answering","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Zero-Shot Learning","url":"/task/zero-shot-learning","datasets_with_task":"/datasets/task/zero-shot-learning"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Visual Question Answering","url":"/task/visual-question-answering-1","datasets_with_task":"/datasets/task/visual-question-answering-1"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Zero-Shot Video Question Answer","url":"/task/zeroshot-video-question-answer","datasets_with_task":"/datasets/task/zeroshot-video-question-answer"},{"name":"Zeroshot Video Question Answer","url":"/task/zeroshot-video-question-answer-1","datasets_with_task":"/datasets/task/zeroshot-video-question-answer-1"}],"languages":[],"variants":["MSVD-QA"],"data_loaders":[],"num_papers_in_archive":61,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-question-answering-on-msvd-qa-1","task":"Visual Question Answering (VQA)","dataset_variant":"MSVD-QA","rows":36,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VLAB","paper":"/paper/vlab-enhancing-video-language-pre-training-by","metrics":{"Accuracy":"0.61"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zeroshot-video-question-answer-on-msvd-qa","task":"Zero-Shot Video Question Answer","dataset_variant":"MSVD-QA","rows":28,"metrics":["Accuracy","Confidence Score"],"first_row_in_archive_order":{"model":"Tarsier (34B)","paper":"/paper/tarsier-recipes-for-training-and-evaluating-1","metrics":{"Accuracy":"80.3","Confidence Score":"4.2"},"code_links":[{"title":"bytedance/tarsier","url":"https://github.com/bytedance/tarsier"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-msvd-qa-2","task":"Visual Question Answering","dataset_variant":"MSVD-QA","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"FrozenBiLM","paper":"/paper/zero-shot-video-question-answering-via-frozen","metrics":{"Accuracy":"0.548"},"code_links":[{"title":"antoyang/FrozenBiLM","url":"https://github.com/antoyang/FrozenBiLM"},{"title":"klauscc/dam","url":"https://github.com/klauscc/dam"},{"title":"sts-vlcc/sts-vlcc","url":"https://github.com/sts-vlcc/sts-vlcc"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-question-answering-on-msvd-qa","task":"Video Question Answering","dataset_variant":"MSVD-QA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"LocVLM-Vid-B","paper":"/paper/learning-to-localize-objects-improves-spatial","metrics":{"Accuracy":"66.1"},"code_links":[{"title":"kahnchana/locvlm","url":"https://github.com/kahnchana/locvlm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-learning-on-msvd-qa","task":"Zero-Shot Learning","dataset_variant":"MSVD-QA","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"HiTeA","paper":"/paper/hitea-hierarchical-temporal-aware-video","metrics":{"Accuracy":"37.4"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/llava-mini-efficient-image-and-video-large","title":"LLaVA-Mini: Efficient Image and Video Large Multimodal Models with One Vision Token","date":"2025-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":4,"samples_unverified":8,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flash-vstream-memory-based-real-time","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","date":"2024-06-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pllava-parameter-free-llava-extension-from-1","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","date":"2024-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-localize-objects-improves-spatial","title":"Learning to Localize Objects Improves Spatial Reasoning in Visual-LLMs","date":"2024-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/st-llm-large-language-models-are-effective-1","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","date":"2024-03-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/elysium-exploring-object-level-perception-in","title":"Elysium: Exploring Object-level Perception in Videos via MLLM","date":"2024-03-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vila-on-pre-training-for-visual-language","title":"VILA: On Pre-training for Visual Language Models","date":"2023-12-12","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chat-univi-unified-visual-representation","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","date":"2023-11-14","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/moviechat-from-dense-token-to-sparse-memory","title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","date":"2023-07-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sas-video-qa-self-adaptive-sampling-for","title":"Self-Adaptive Sampling for Efficient Video Question-Answering on Image--Text Models","date":"2023-07-09","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/lightweight-recurrent-cross-modal-encoder-for","title":"Lightweight Recurrent Cross-modal Encoder for Video Question Answering","date":"2023-06-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-chatgpt-towards-detailed-video","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","date":"2023-06-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":18,"samples_unverified":7,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vlab-enhancing-video-language-pre-training-by","title":"VLAB: Enhancing Video Language Pre-training by Feature Adapting and Blending","date":"2023-05-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/videochat-chat-centric-video-understanding","title":"VideoChat: Chat-Centric Video Understanding","date":"2023-05-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/llama-adapter-v2-parameter-efficient-visual","title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","date":"2023-04-28","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-efficient-video-and-language","title":"MuLTI: Efficient Video-and-Language Understanding with Text-Guided MultiWay-Sampler and Multiple Choice Modeling","date":"2023-03-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-question-answering-with-iterative-video","title":"Video Question Answering with Iterative Video-Text Co-Tokenization","date":"2022-08-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":14,"samples_unverified":20,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":0,"samples_unverified":21,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/all-in-one-exploring-unified-video-language","title":"All in One: Exploring Unified Video-Language Pre-training","date":"2022-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/align-and-prompt-video-and-language-pre","title":"Align and Prompt: Video-and-Language Pre-training with Entity Prompts","date":"2021-12-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dualvgr-a-dual-visual-graph-reasoning-unit","title":"DualVGR: A Dual-Visual Graph Reasoning Unit for Video Question Answering","date":"2021-07-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/just-ask-learning-to-answer-questions-from","title":"Just Ask: Learning to Answer Questions from Millions of Narrated Videos","date":"2020-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/noise-estimation-using-density-estimation-for","title":"Noise Estimation Using Density Estimation for Self-Supervised Multimodal Learning","date":"2020-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-conditional-relation-networks","title":"Hierarchical Conditional Relation Networks for Video Question Answering","date":"2020-02-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/heterogeneous-memory-enhanced-multimodal","title":"Heterogeneous Memory Enhanced Multimodal Attention Model for Video Question Answering","date":"2019-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/motion-appearance-co-memory-networks-for","title":"Motion-Appearance Co-Memory Networks for Video Question Answering","date":"2018-03-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/tgif-qa-toward-spatio-temporal-reasoning-in","title":"TGIF-QA: Toward Spatio-Temporal Reasoning in Visual Question Answering","date":"2017-04-14","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":32,"samples_harvested":294,"samples_ran":144,"samples_unverified":150,"pointer_only_for_licence":74,"papers_with_no_sample_that_ran":4,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}