{"url":"/dataset/next-qa-open-ended-videoqa","name":"NExT-QA (Open-ended VideoQA)","full_name":null,"description_markdown":"NExT-QA is a VideoQA benchmark targeting the explanation of video contents. It challenges QA models to reason about the causal and temporal actions and understand the rich object interactions in daily activities.  This page records LLMs for answer evaluation.","description_withheld":null,"homepage":"https://github.com/doc-doc/NExT-OE","introduced_date":"2021-05-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/next-qa-next-phase-of-question-answering-to","title":"NExT-QA:Next Phase of Question-Answering to Explaining Temporal Actions","first_author":"Junbin Xiao","url":null},"license":{"name":"MIT","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["NExT-QA (Open-ended VideoQA)"],"data_loaders":[],"num_papers_in_archive":7,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-next-qa-open-ended","task":"Question Answering","dataset_variant":"NExT-QA (Open-ended VideoQA)","rows":6,"metrics":["Accuracy","Confidence Score"],"first_row_in_archive_order":{"model":"Flash-VStream","paper":"/paper/flash-vstream-memory-based-real-time","metrics":{"Accuracy":"61.6","Confidence Score":"3.4"},"code_links":[{"title":"IVGSZ/Flash-VStream","url":"https://github.com/IVGSZ/Flash-VStream"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/flash-vstream-memory-based-real-time","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","date":"2024-06-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/moviechat-question-aware-sparse-memory-for","title":"MovieChat+: Question-aware Sparse Memory for Long Video Question Answering","date":"2024-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":11,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vista-llama-reliable-video-narrator-via-equal","title":"Vista-LLaMA: Reliable Video Narrator via Equal Distance to Visual Tokens","date":"2023-12-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/moviechat-from-dense-token-to-sparse-memory","title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","date":"2023-07-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/video-chatgpt-towards-detailed-video","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","date":"2023-06-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/videochat-chat-centric-video-understanding","title":"VideoChat: Chat-Centric Video Understanding","date":"2023-05-10","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":14,"samples_ran":11,"samples_unverified":3,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}