{"url":"/dataset/next-qa","name":"NExT-QA","full_name":null,"description_markdown":"**NExT-QA** is a VideoQA benchmark targeting the explanation of video contents. It challenges QA models to reason about the causal and temporal actions and understand the rich object interactions in daily activities, e.g., \"why is the boy crying?\" and \"How does the lady react after the boy fall backward?\". It supports both multi-choice and generative open-ended QA tasks. The videos are untrimmed and the questions usually invoke local video contents for answers.","description_withheld":null,"homepage":"https://github.com/doc-doc/NExT-QA","introduced_date":"2021-06-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/next-qa-next-phase-of-question-answering-to-1","title":"NExT-QA: Next Phase of Question-Answering to Explaining Temporal Actions","first_author":"Junbin Xiao","url":null},"license":{"name":"MIT","url":"https://github.com/doc-doc/NExT-QA/blob/main/LICENSE"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Actions","url":"/datasets/modality/actions"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Zero-Shot Video Question Answer","url":"/task/zeroshot-video-question-answer","datasets_with_task":"/datasets/task/zeroshot-video-question-answer"},{"name":"Temporal/Casual QA","url":"/task/temporal-casual-qa","datasets_with_task":"/datasets/task/temporal-casual-qa"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["NExT-QA","NExT-GQA","NExT-QA (Open-ended VideoQA)"],"data_loaders":[],"num_papers_in_archive":174,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-next-qa","task":"Video Question Answering","dataset_variant":"NExT-QA","rows":47,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"LinVT-Qwen2-VL\n(7B)","paper":"/paper/linvt-empower-your-image-level-large-language","metrics":{"Accuracy":"85.5"},"code_links":[{"title":"gls0425/linvt","url":"https://github.com/gls0425/linvt"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-question-answer-on-next-qa","task":"Zero-Shot Video Question Answer","dataset_variant":"NExT-QA","rows":27,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"VideoMultiAgent (GPT-4o)","paper":"/paper/videomultiagents-a-multi-agent-framework-for","metrics":{"Accuracy":"79.6"},"code_links":[{"title":"panasonicconnect/videomultiagents","url":"https://github.com/panasonicconnect/videomultiagents"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/temporal-casual-qa-on-next-qa","task":"Temporal/Casual QA","dataset_variant":"NExT-QA","rows":8,"metrics":["WUPS"],"first_row_in_archive_order":{"model":"PaLI-X","paper":"/paper/pali-x-on-scaling-up-a-multilingual-vision","metrics":{"WUPS":"38.3"},"code_links":[{"title":"kyegomez/PALI","url":"https://github.com/kyegomez/PALI"},{"title":"doc-doc/NExT-OE","url":"https://github.com/doc-doc/NExT-OE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/videomultiagents-a-multi-agent-framework-for","title":"VideoMultiAgents: A Multi-Agent Framework for Video Question Answering","date":"2025-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/perceptionlm-open-access-data-and-models-for","title":"PerceptionLM: Open-Access Data and Models for Detailed Visual Understanding","date":"2025-04-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/agentic-keyframe-search-for-video-question","title":"Agentic Keyframe Search for Video Question Answering","date":"2025-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bimba-selective-scan-compression-for-long","title":"BIMBA: Selective-Scan Compression for Long-Range Video Question Answering","date":"2025-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/enter-event-based-interpretable-reasoning-for","title":"ENTER: Event Based Interpretable Reasoning for VideoQA","date":"2025-01-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/videollama-3-frontier-multimodal-foundation","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","date":"2025-01-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":5,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vidctx-context-aware-video-question-answering","title":"VidCtx: Context-aware Video Question Answering with Image Models","date":"2024-12-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":4,"samples_unverified":8,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/nvila-efficient-frontier-visual-language","title":"NVILA: Efficient Frontier Visual Language Models","date":"2024-12-05","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-instruction-tuning-with-synthetic-data","title":"Video Instruction Tuning With Synthetic Data","date":"2024-10-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/oryx-mllm-on-demand-spatial-temporal","title":"Oryx MLLM: On-Demand Spatial-Temporal Understanding at Arbitrary Resolution","date":"2024-09-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","date":"2024-09-18","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":8,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/longvila-scaling-long-context-visual-language","title":"LongVILA: Scaling Long-Context Visual Language Models for Long Videos","date":"2024-08-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mplug-owl3-towards-long-image-sequence","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","date":"2024-08-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/llava-onevision-easy-visual-task-transfer","title":"LLaVA-OneVision: Easy Visual Task Transfer","date":"2024-08-06","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llava-next-interleave-tackling-multi-image","title":"LLaVA-NeXT-Interleave: Tackling Multi-image, Video, and 3D in Large Multimodal Models","date":"2024-07-10","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/long-context-transfer-from-language-to-vision","title":"Long Context Transfer from Language to Vision","date":"2024-06-24","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/too-many-frames-not-all-useful-efficient","title":"Too Many Frames, Not All Useful: Efficient Strategies for Long-Form Video QA","date":"2024-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deepstack-deeply-stacking-visual-tokens-is","title":"DeepStack: Deeply Stacking Visual Tokens is Surprisingly Simple and Effective for LMMs","date":"2024-06-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/videotree-adaptive-tree-based-video","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","date":"2024-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/morevqa-exploring-modular-reasoning-models","title":"MoReVQA: Exploring Modular Reasoning Models for Video Question Answering","date":"2024-04-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/traveler-a-multi-lmm-agent-framework-for","title":"TraveLER: A Modular Multi-LMM Agent Framework for Video Question-Answering","date":"2024-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/understanding-long-videos-in-one-multimodal","title":"Understanding Long Videos with Multimodal Language Models","date":"2024-03-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/language-repository-for-long-video","title":"Language Repository for Long Video Understanding","date":"2024-03-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":9,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videoagent-long-form-video-understanding-with","title":"VideoAgent: Long-form Video Understanding with Large Language Model as Agent","date":"2024-03-15","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/lstp-language-guided-spatial-temporal-prompt","title":"Efficient Temporal Extrapolation of Multimodal Large Language Models with Temporal Grounding Bridge","date":"2024-02-25","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/question-instructed-visual-descriptions-for","title":"Question-Instructed Visual Descriptions for Zero-Shot Video Question Answering","date":"2024-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/crema-multimodal-compositional-video","title":"CREMA: Generalizable and Efficient Video-Language Reasoning via Multimodal Modular Fusion","date":"2024-02-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glance-and-focus-memory-prompting-for-multi-1","title":"Glance and Focus: Memory Prompting for Multi-Event Video Question Answering","date":"2024-01-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":9,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-simple-llm-framework-for-long-range-video","title":"A Simple LLM Framework for Long-Range Video Question-Answering","date":"2023-12-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-conditioned-resampler-for-long-form","title":"Text-Conditioned Resampler For Long Form Video Understanding","date":"2023-12-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-video-question-answering-with","title":"Zero-Shot Video Question Answering with Procedural Programs","date":"2023-12-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/rtq-rethinking-video-language-understanding","title":"RTQ: Rethinking Video-language Understanding Based on Image-text Model","date":"2023-12-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vamos-versatile-action-models-for-video","title":"Vamos: Versatile Action Models for Video Understanding","date":"2023-11-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mirasol3b-a-multimodal-autoregressive-model","title":"Mirasol3B: A Multimodal Autoregressive model for time-aligned and contextual modalities","date":"2023-11-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-temporal-and-causal","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","date":"2023-10-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pali-3-vision-language-models-smaller-faster","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","date":"2023-10-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/atm-action-temporality-modeling-for-video","title":"ATM: Action Temporality Modeling for Video Question Answering","date":"2023-09-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/generative-pretraining-in-multimodality","title":"Emu: Generative Pretraining in Multimodality","date":"2023-07-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/retrieving-to-answer-zero-shot-video-question","title":"Retrieving-to-Answer: Zero-Shot Video Question Answering with Frozen Large Language Models","date":"2023-06-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/paxion-patching-action-knowledge-in-video-1","title":"Paxion: Patching Action Knowledge in Video-Language Foundation Models","date":"2023-05-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-chained-image-language-model-for-video-1","title":"Self-Chained Image-Language Model for Video Localization and Question Answering","date":"2023-05-11","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/verbs-in-action-improving-verb-understanding","title":"Verbs in Action: Improving verb understanding in video-language models","date":"2023-04-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vipergpt-visual-inference-via-python","title":"ViperGPT: Visual Inference via Python Execution for Reasoning","date":"2023-03-14","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/contrastive-video-question-answering-via","title":"Contrastive Video Question Answering via Video Graph Transformer","date":"2023-02-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/semi-parametric-video-grounded-text","title":"Semi-Parametric Video-Grounded Text Generation","date":"2023-01-27","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mist-multi-modal-iterative-spatial-temporal","title":"MIST: Multi-modal Iterative Spatial-Temporal Transformer for Long-form Video Question Answering","date":"2022-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-graph-transformer-for-video-question","title":"Video Graph Transformer for Video Question Answering","date":"2022-07-12","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":9,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-the-video-in-video-language","title":"Revisiting the \"Video\" in Video-Language Understanding","date":"2022-06-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":18,"samples_unverified":6,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/2-5-1-d-spatio-temporal-scene-graphs-for","title":"(2.5+1)D Spatio-Temporal Scene Graphs for Video Question Answering","date":"2022-02-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/video-as-conditional-graph-hierarchy-for","title":"Video as Conditional Graph Hierarchy for Multi-Granular Question Answering","date":"2021-12-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":37,"samples_harvested":289,"samples_ran":181,"samples_unverified":108,"pointer_only_for_licence":48,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}