{"url":"/dataset/videoinstruct","name":"VideoInstruct","full_name":"Video Instruction Dataset","description_markdown":"Video Instruction Dataset is used to train Video-ChatGPT. It consists of 100,000 high-quality video instruction pairs. employs a combination of human-assisted and semi-automatic annotation techniques, aiming to produce high-quality video instruction data. These methods create question-answer pairs related to\r\n\r\n1. Video summarization\r\n2. Description-based question-answers (exploring spatial, temporal, relationships, and reasoning concepts)\r\n3. Creative/generative question-answers","description_withheld":null,"homepage":"https://mbzuai-oryx.github.io/Video-ChatGPT","introduced_date":"2023-06-08","introduced_date_note":null,"introduced_by":{"paper":"/paper/video-chatgpt-towards-detailed-video","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","first_author":"Muhammad Maaz","url":null},"license":{"name":"Creative Commons Attribution 4.0","url":"http://creativecommons.org/licenses/by-nc-sa/4.0"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Question Answering","url":"/task/video-question-answering","datasets_with_task":"/datasets/task/video-question-answering"},{"name":"Video-based Generative Performance Benchmarking","url":"/task/video-based-generative-performance","datasets_with_task":"/datasets/task/video-based-generative-performance"},{"name":"Video-based Generative Performance Benchmarking (Correctness of Information)","url":"/task/video-based-generative-performance-1","datasets_with_task":"/datasets/task/video-based-generative-performance-1"},{"name":"Video-based Generative Performance Benchmarking (Detail Orientation))","url":"/task/video-based-generative-performance-2","datasets_with_task":"/datasets/task/video-based-generative-performance-2"},{"name":"Video-based Generative Performance Benchmarking (Contextual Understanding)","url":"/task/video-based-generative-performance-3","datasets_with_task":"/datasets/task/video-based-generative-performance-3"},{"name":"Video-based Generative Performance Benchmarking (Temporal Understanding)","url":"/task/video-based-generative-performance-4","datasets_with_task":"/datasets/task/video-based-generative-performance-4"},{"name":"Video-based Generative Performance Benchmarking (Consistency)","url":"/task/video-based-generative-performance-5","datasets_with_task":"/datasets/task/video-based-generative-performance-5"},{"name":"VCGBench-Diverse","url":"/task/vcgbench-diverse","datasets_with_task":"/datasets/task/vcgbench-diverse"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["VideoInstruct"],"data_loaders":[],"num_papers_in_archive":30,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-based-generative-performance","task":"Video-based Generative Performance Benchmarking","dataset_variant":"VideoInstruct","rows":23,"metrics":["mean","Correctness of Information","Detail Orientation","Contextual Understanding","Temporal Understanding","Consistency"],"first_row_in_archive_order":{"model":"PPLLaVA-7B-dpo","paper":"/paper/ppllava-varied-video-sequence-understanding","metrics":{"Consistency":"3.81","Contextual Understanding":"4.21","Correctness of Information":"3.85","Detail Orientation":"3.56","Temporal Understanding":"3.21","mean":"3.73"},"code_links":[{"title":"farewellthree/ppllava","url":"https://github.com/farewellthree/ppllava"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-based-generative-performance-1","task":"Video-based Generative Performance Benchmarking (Correctness of Information)","dataset_variant":"VideoInstruct","rows":18,"metrics":["gpt-score"],"first_row_in_archive_order":{"model":"PPLLaVA-7B","paper":"/paper/ppllava-varied-video-sequence-understanding","metrics":{"gpt-score":"3.85"},"code_links":[{"title":"farewellthree/ppllava","url":"https://github.com/farewellthree/ppllava"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-based-generative-performance-2","task":"Video-based Generative Performance Benchmarking (Consistency)","dataset_variant":"VideoInstruct","rows":18,"metrics":["gpt-score"],"first_row_in_archive_order":{"model":"PPLLaVA-7B","paper":"/paper/ppllava-varied-video-sequence-understanding","metrics":{"gpt-score":"3.81"},"code_links":[{"title":"farewellthree/ppllava","url":"https://github.com/farewellthree/ppllava"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-based-generative-performance-3","task":"Video-based Generative Performance Benchmarking (Contextual Understanding)","dataset_variant":"VideoInstruct","rows":18,"metrics":["gpt-score"],"first_row_in_archive_order":{"model":"PPLLaVA-7B","paper":"/paper/ppllava-varied-video-sequence-understanding","metrics":{"gpt-score":"4.21"},"code_links":[{"title":"farewellthree/ppllava","url":"https://github.com/farewellthree/ppllava"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-based-generative-performance-4","task":"Video-based Generative Performance Benchmarking (Detail Orientation))","dataset_variant":"VideoInstruct","rows":18,"metrics":["gpt-score"],"first_row_in_archive_order":{"model":"PPLLaVA-7B","paper":"/paper/ppllava-varied-video-sequence-understanding","metrics":{"gpt-score":"3.56"},"code_links":[{"title":"farewellthree/ppllava","url":"https://github.com/farewellthree/ppllava"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-based-generative-performance-5","task":"Video-based Generative Performance Benchmarking (Temporal Understanding)","dataset_variant":"VideoInstruct","rows":18,"metrics":["gpt-score"],"first_row_in_archive_order":{"model":"PPLLaVA-7B","paper":"/paper/ppllava-varied-video-sequence-understanding","metrics":{"gpt-score":"3.21"},"code_links":[{"title":"farewellthree/ppllava","url":"https://github.com/farewellthree/ppllava"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/vcgbench-diverse-on-videoinstruct","task":"VCGBench-Diverse","dataset_variant":"VideoInstruct","rows":6,"metrics":["mean","Correctness of Information","Detail Orientation","Contextual Understanding","Temporal Understanding","Consistency","Dense Captioning","Spatial Understanding","Reasoning"],"first_row_in_archive_order":{"model":"VideoGPT+","paper":"/paper/videogpt-integrating-image-and-video-encoders","metrics":{"Consistency":"2.59","Contextual Understanding":"2.81","Correctness of Information":"2.46","Dense Captioning":"1.38","Detail Orientation":"2.73","Reasoning":"3.63","Spatial Understanding":"2.80","Temporal Understanding":"1.78","mean":"2.47"},"code_links":[{"title":"mbzuai-oryx/videogpt-plus","url":"https://github.com/mbzuai-oryx/videogpt-plus"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/ts-llava-constructing-visual-tokens-through","title":"TS-LLaVA: Constructing Visual Tokens through Thumbnail-and-Sampling for Training-Free Video Large Language Models","date":"2024-11-17","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pllava-parameter-free-llava-extension-from-1","title":"PLLaVA : Parameter-free LLaVA Extension from Images to Videos for Video Dense Captioning","date":"2024-04-25","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","rows_on_this_dataset":5,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/st-llm-large-language-models-are-effective-1","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","date":"2024-03-30","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lita-language-instructed-temporal","title":"LITA: Language Instructed Temporal-Localization Assistant","date":"2024-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":6,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tuning-large-multimodal-models-for-videos","title":"Tuning Large Multimodal Models for Videos using Reinforcement Learning from AI Feedback","date":"2024-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","rows_on_this_dataset":13,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":7,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chat-univi-unified-visual-representation","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","date":"2023-11-14","rows_on_this_dataset":7,"code_links":4,"syntology":null},{"paper":"/paper/one-for-all-video-conversation-is-feasible","title":"BT-Adapter: Video Conversation is Feasible Without Video Instruction Tuning","date":"2023-09-27","rows_on_this_dataset":13,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/moviechat-from-dense-token-to-sparse-memory","title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","date":"2023-07-31","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/video-chatgpt-towards-detailed-video","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","date":"2023-06-08","rows_on_this_dataset":7,"code_links":2,"syntology":null},{"paper":"/paper/video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","rows_on_this_dataset":6,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":18,"samples_unverified":7,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videochat-chat-centric-video-understanding","title":"VideoChat: Chat-Centric Video Understanding","date":"2023-05-10","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/llama-adapter-v2-parameter-efficient-visual","title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","date":"2023-04-28","rows_on_this_dataset":6,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":16,"samples_harvested":121,"samples_ran":71,"samples_unverified":50,"pointer_only_for_licence":45,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}