{"url":"/dataset/ovbench","name":"OVBench","full_name":null,"description_markdown":"OVBench is a benchmark tailored for **real-time video understanding**:\r\n\r\n- **Memory, Perception, and Prediction of Temporal Contexts**: Questions are framed to reference the present state of entities, requiring models to memorize/perceive/predict past/present/future temporal contexts over time.\r\n- **Dynamic Spatio-temporal Interaction**: The benchmark demands precise real-time interactions with video content, where actions, objects, and events must be understood in the context of their spatial and temporal relationships.\r\n- **Contextual Awareness at Specific Moments**: Real-time questions are contextual, changing based on the specific timestamp they are asked, requiring a deep understanding of how temporal context evolves.","description_withheld":null,"homepage":"https://videochat-online.github.io/","introduced_date":"2024-12-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/online-video-understanding-a-comprehensive","title":"Online Video Understanding: OVBench and VideoChat-Online","first_author":"Zhenpeng Huang","url":null},"license":null,"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Video Understanding","url":"/task/video-understanding","datasets_with_task":"/datasets/task/video-understanding"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["OVBench"],"data_loaders":[],"num_papers_in_archive":14,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-question-answering-on-ovbench","task":"Video Question Answering","dataset_variant":"OVBench","rows":16,"metrics":["AVG"],"first_row_in_archive_order":{"model":"Seed1.5-VL","paper":"/paper/seed1-5-vl-technical-report","metrics":{"AVG":"60.0"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/seed1-5-vl-technical-report","title":"Seed1.5-VL Technical Report","date":"2025-05-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/online-video-understanding-a-comprehensive","title":"Online Video Understanding: OVBench and VideoChat-Online","date":"2024-12-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","date":"2024-09-18","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":8,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llava-onevision-easy-visual-task-transfer","title":"LLaVA-OneVision: Easy Visual Task Transfer","date":"2024-08-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/long-context-transfer-from-language-to-vision","title":"Long Context Transfer from Language to Vision","date":"2024-06-24","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videollm-online-online-video-large-language-1","title":"VideoLLM-online: Online Video Large Language Model for Streaming Video","date":"2024-06-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/flash-vstream-memory-based-real-time","title":"Flash-VStream: Memory-Based Real-Time Understanding for Long Video Streams","date":"2024-06-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lita-language-instructed-temporal","title":"LITA: Language Instructed Temporal-Localization Assistant","date":"2024-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/moviechat-from-dense-token-to-sparse-memory","title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","date":"2023-07-31","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":54,"samples_ran":31,"samples_unverified":23,"pointer_only_for_licence":18,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}