{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/video-chatgpt-towards-detailed-video","title":"Video-ChatGPT: Towards Detailed Video Understanding via Large Vision and Language Models","arxiv_id":"2306.05424","date":"2023-06-08","proceeding":null,"authors":["Muhammad Maaz","Hanoona Rasheed","Salman Khan","Fahad Shahbaz Khan"],"abstract":"Conversation agents fueled by Large Language Models (LLMs) are providing a new way to interact with visual data. While there have been initial attempts for image-based conversation models, this work addresses the under-explored field of \\emph{video-based conversation} by introducing Video-ChatGPT. It is a multimodal model that merges a video-adapted visual encoder with an LLM. The resulting model is capable of understanding and generating detailed conversations about videos. We introduce a new dataset of 100,000 video-instruction pairs used to train Video-ChatGPT acquired via manual and semi-automated pipeline that is easily scalable and robust to label noise. We also develop a quantitative evaluation framework for video-based dialogue models to objectively analyze the strengths and weaknesses of video-based dialogue models. Code: https://github.com/mbzuai-oryx/Video-ChatGPT.","url_abs":"https://arxiv.org/abs/2306.05424v2","url_pdf":"https://arxiv.org/pdf/2306.05424v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"video-chatgpt-towards-detailed-video","repo_url":"https://github.com/mbzuai-oryx/video-chatgpt","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"CC-BY-4.0"}},{"paper_slug":"video-chatgpt-towards-detailed-video","repo_url":"https://github.com/qiujihao19/artemis","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"question-answering","task_name":"Question Answering"},{"task_slug":"vcgbench-diverse","task_name":"VCGBench-Diverse"},{"task_slug":"video-question-answering","task_name":"Video Question Answering"},{"task_slug":"video-understanding","task_name":"Video Understanding"},{"task_slug":"video-based-generative-performance","task_name":"Video-based Generative Performance Benchmarking"},{"task_slug":"video-based-generative-performance-5","task_name":"Video-based Generative Performance Benchmarking (Consistency)"},{"task_slug":"video-based-generative-performance-3","task_name":"Video-based Generative Performance Benchmarking (Contextual Understanding)"},{"task_slug":"video-based-generative-performance-1","task_name":"Video-based Generative Performance Benchmarking (Correctness of Information)"},{"task_slug":"video-based-generative-performance-2","task_name":"Video-based Generative Performance Benchmarking (Detail Orientation))"},{"task_slug":"video-based-generative-performance-4","task_name":"Video-based Generative Performance Benchmarking (Temporal Understanding)"},{"task_slug":"zeroshot-video-question-answer","task_name":"Zero-Shot Video Question Answer"}],"methods":[],"datasets_introduced":[{"slug":"videoinstruct","name":"VideoInstruct","full_name":"Video Instruction Dataset"}],"methods_introduced":[],"results":[{"leaderboard":"/sota/question-answering-on-next-qa-open-ended","task":"Question Answering","dataset":"NExT-QA (Open-ended VideoQA)","model":"Video-ChatGPT","rank_in_archive_order":5,"of":6,"metrics":{"Accuracy":"54.6","Confidence Score":"3.2"},"uses_additional_data":false},{"leaderboard":"/sota/vcgbench-diverse-on-videoinstruct","task":"VCGBench-Diverse","dataset":"VideoInstruct","model":"Video-ChatGPT","rank_in_archive_order":6,"of":6,"metrics":{"Consistency":"2.06","Contextual Understanding":"2.46","Correctness of Information":"2.07","Dense Captioning":"0.89","Detail Orientation":"2.42","Reasoning":"3.60","Spatial Understanding":"2.25","Temporal Understanding":"1.39","mean":"2.08"},"uses_additional_data":false},{"leaderboard":"/sota/video-question-answering-on-activitynet-qa","task":"Video Question Answering","dataset":"ActivityNet-QA","model":"Video-ChatGPT","rank_in_archive_order":29,"of":36,"metrics":{"Accuracy":"35.2","Confidence score":"2.7"},"uses_additional_data":false},{"leaderboard":"/sota/video-question-answering-on-mvbench","task":"Video Question Answering","dataset":"MVBench","model":"Video-ChatGPT","rank_in_archive_order":20,"of":22,"metrics":{"Avg.":"32.7"},"uses_additional_data":false},{"leaderboard":"/sota/video-based-generative-performance","task":"Video-based Generative Performance Benchmarking","dataset":"VideoInstruct","model":"Video-ChatGPT","rank_in_archive_order":20,"of":23,"metrics":{"Consistency":"2.37","Contextual Understanding":"2.62","Correctness of Information":"2.4","Detail Orientation":"2.52","Temporal Understanding":"1.98","mean":"2.38"},"uses_additional_data":false},{"leaderboard":"/sota/video-based-generative-performance-2","task":"Video-based Generative Performance Benchmarking (Consistency)","dataset":"VideoInstruct","model":"Video-ChatGPT","rank_in_archive_order":14,"of":18,"metrics":{"gpt-score":"2.37"},"uses_additional_data":false},{"leaderboard":"/sota/video-based-generative-performance-3","task":"Video-based Generative Performance Benchmarking (Contextual Understanding)","dataset":"VideoInstruct","model":"Video-ChatGPT","rank_in_archive_order":15,"of":18,"metrics":{"gpt-score":"2.62"},"uses_additional_data":false},{"leaderboard":"/sota/video-based-generative-performance-1","task":"Video-based Generative Performance Benchmarking (Correctness of Information)","dataset":"VideoInstruct","model":"Video-ChatGPT","rank_in_archive_order":14,"of":18,"metrics":{"gpt-score":"2.40"},"uses_additional_data":false},{"leaderboard":"/sota/video-based-generative-performance-4","task":"Video-based Generative Performance Benchmarking (Detail Orientation))","dataset":"VideoInstruct","model":"Video-ChatGPT","rank_in_archive_order":14,"of":18,"metrics":{"gpt-score":"2.52"},"uses_additional_data":false},{"leaderboard":"/sota/video-based-generative-performance-5","task":"Video-based Generative Performance Benchmarking (Temporal Understanding)","dataset":"VideoInstruct","model":"Video-ChatGPT","rank_in_archive_order":15,"of":18,"metrics":{"gpt-score":"1.98"},"uses_additional_data":false},{"leaderboard":"/sota/zeroshot-video-question-answer-on-activitynet","task":"Zero-Shot Video Question Answer","dataset":"ActivityNet-QA","model":"Video-ChatGPT","rank_in_archive_order":24,"of":28,"metrics":{"Accuracy":"35.2","Confidence Score":"2.7"},"uses_additional_data":false},{"leaderboard":"/sota/zeroshot-video-question-answer-on-msrvtt-qa","task":"Zero-Shot Video Question Answer","dataset":"MSRVTT-QA","model":"Video-ChatGPT-7B","rank_in_archive_order":27,"of":30,"metrics":{"Accuracy":"49.3","Confidence Score":"2.8"},"uses_additional_data":false},{"leaderboard":"/sota/zeroshot-video-question-answer-on-msvd-qa","task":"Zero-Shot Video Question Answer","dataset":"MSVD-QA","model":"Video-ChatGPT-7B","rank_in_archive_order":24,"of":28,"metrics":{"Accuracy":"64.9","Confidence Score":"3.3"},"uses_additional_data":false},{"leaderboard":"/sota/zeroshot-video-question-answer-on-tgif-qa","task":"Zero-Shot Video Question Answer","dataset":"TGIF-QA","model":"Video-ChatGPT-7B","rank_in_archive_order":12,"of":14,"metrics":{"Accuracy":"51.4","Confidence Score":"3.0"},"uses_additional_data":false},{"leaderboard":"/sota/zero-shot-video-question-answer-on-vnbench","task":"Zero-Shot Video Question Answer","dataset":"VNBench","model":"VideoChatGPT","rank_in_archive_order":9,"of":9,"metrics":{"Accuracy":"4.1"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2306.05424","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}