{"url":"/dataset/egoschema","name":"EgoSchema","full_name":null,"description_markdown":"**EgoSchema** is very long-form video question-answering dataset, and benchmark to evaluate long video understanding capabilities of modern vision and language systems. Derived from Ego4D, EgoSchema consists of over 5000 human curated multiple choice question answer pairs, spanning over 250 hours of real video data, covering a very broad range of natural human activity and behavior.","description_withheld":null,"homepage":"https://egoschema.github.io/","introduced_date":"2023-08-17","introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Zero-Shot Video Question Answer","url":"/task/zeroshot-video-question-answer","datasets_with_task":"/datasets/task/zeroshot-video-question-answer"}],"languages":[],"variants":["EgoSchema","EgoSchema (subset)","EgoSchema (fullset)"],"data_loaders":[],"num_papers_in_archive":112,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/zero-shot-video-question-answer-on-egoschema-1","task":"Zero-Shot Video Question Answer","dataset_variant":"EgoSchema (fullset)","rows":29,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"BIMBA-LLaVA-Qwen2-7B","paper":"/paper/bimba-selective-scan-compression-for-long","metrics":{"Accuracy":"71.14"},"code_links":[{"title":"md-mohaiminul/BIMBA","url":"https://github.com/md-mohaiminul/BIMBA"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-video-question-answer-on-egoschema","task":"Zero-Shot Video Question Answer","dataset_variant":"EgoSchema (subset)","rows":14,"metrics":["Accuracy","Inference Speed (s)"],"first_row_in_archive_order":{"model":"Tarsier (34B)","paper":"/paper/tarsier-recipes-for-training-and-evaluating-1","metrics":{"Accuracy":"68.6"},"code_links":[{"title":"bytedance/tarsier","url":"https://github.com/bytedance/tarsier"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-vqa-on-egoschema","task":"Visual Question Answering (VQA)","dataset_variant":"EgoSchema","rows":1,"metrics":["Acc"],"first_row_in_archive_order":{"model":"Lyra-Pro","paper":"/paper/lyra-an-efficient-and-speech-centric","metrics":{"Acc":"75.8"},"code_links":[{"title":"dvlab-research/Lyra","url":"https://github.com/dvlab-research/Lyra"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/qwen2-5-omni-technical-report","title":"Qwen2.5-Omni Technical Report","date":"2025-03-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bimba-selective-scan-compression-for-long","title":"BIMBA: Selective-Scan Compression for Long-Range Video Question Answering","date":"2025-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":3,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":4,"samples_unverified":8,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-rag-visually-aligned-retrieval","title":"Video-RAG: Visually-aligned Retrieval-Augmented Long Video Comprehension","date":"2024-11-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/longvu-spatiotemporal-adaptive-compression","title":"LongVU: Spatiotemporal Adaptive Compression for Long Video-Language Understanding","date":"2024-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/too-many-frames-not-all-useful-efficient","title":"Too Many Frames, Not All Useful: Efficient Strategies for Long-Form Video QA","date":"2024-06-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videotree-adaptive-tree-based-video","title":"VideoTree: Adaptive Tree-based Video Representation for LLM Reasoning on Long Videos","date":"2024-05-29","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/traveler-a-multi-lmm-agent-framework-for","title":"TraveLER: A Modular Multi-LMM Agent Framework for Video Question-Answering","date":"2024-04-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/understanding-long-videos-in-one-multimodal","title":"Understanding Long Videos with Multimodal Language Models","date":"2024-03-25","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/language-repository-for-long-video","title":"Language Repository for Long Video Understanding","date":"2024-03-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":9,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/video-recap-recursive-captioning-of-hour-long","title":"Video ReCap: Recursive Captioning of Hour-Long Videos","date":"2024-02-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-simple-llm-framework-for-long-range-video","title":"A Simple LLM Framework for Long-Range Video Question-Answering","date":"2023-12-28","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":5,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vamos-versatile-action-models-for-video","title":"Vamos: Versatile Action Models for Video Understanding","date":"2023-11-22","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":7,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-chained-image-language-model-for-video-1","title":"Self-Chained Image-Language Model for Video Localization and Question Answering","date":"2023-05-11","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mplug-owl-modularization-empowers-large","title":"mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality","date":"2023-04-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":14,"samples_unverified":20,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":21,"samples_harvested":194,"samples_ran":115,"samples_unverified":79,"pointer_only_for_licence":32,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}