{"url":"/dataset/vidchapters-7m","name":"VidChapters-7M","full_name":null,"description_markdown":"VidChapters-7M is a dataset of 817K user-chaptered videos including 7M chapters in total. VidChapters-7M is automatically created from videos online in a scalable manner by scraping user-annotated chapters and hence without any additional manual annotation. It is designed for training and evaluating models for video chapter generation with or without ground-truth boundaries, and video chapter grounding, as well as for video-language pretraining.","description_withheld":null,"homepage":"https://antoyang.github.io/vidchapters.html","introduced_date":"2023-09-25","introduced_date_note":null,"introduced_by":null,"license":{"name":"MIT","url":"https://github.com/antoyang/VidChapters/blob/main/LICENSE"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Captioning","url":"/task/video-captioning","datasets_with_task":"/datasets/task/video-captioning"},{"name":"Dense Video Captioning","url":"/task/dense-video-captioning","datasets_with_task":"/datasets/task/dense-video-captioning"},{"name":"Language-Based Temporal Localization","url":"/task/language-based-temporal-localization","datasets_with_task":"/datasets/task/language-based-temporal-localization"},{"name":"Video Chaptering","url":"/task/video-chaptering","datasets_with_task":"/datasets/task/video-chaptering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["VidChapters-7M"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/language-based-temporal-localization-on","task":"Language-Based Temporal Localization","dataset_variant":"VidChapters-7M","rows":2,"metrics":["R1@.9","R@10s"],"first_row_in_archive_order":{"model":"ReVisionLLM","paper":"/paper/revisionllm-recursive-vision-language-model","metrics":{"R1@.9":"15.2"},"code_links":[{"title":"tanveer81/revisionllm","url":"https://github.com/tanveer81/revisionllm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-chaptering-on-vidchapters-7m","task":"Video Chaptering","dataset_variant":"VidChapters-7M","rows":2,"metrics":["P@5s","CIDEr","P@0.5","P@0.7","P@3s","R@0.5","R@0.7","R@3s","R@5s","SODA"],"first_row_in_archive_order":{"model":"Chapter-Llama","paper":"/paper/chapter-llama-efficient-chaptering-in-hour-1","metrics":{"P@5s":"52.0"},"code_links":[{"title":"lucas-ventura/chapter-llama","url":"https://github.com/lucas-ventura/chapter-llama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dense-video-captioning-on-vidchapters-7m","task":"Dense Video Captioning","dataset_variant":"VidChapters-7M","rows":1,"metrics":["CIDEr"],"first_row_in_archive_order":{"model":"Vid2Seq","paper":null,"metrics":{"CIDEr":"55.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-captioning-on-vidchapters-7m","task":"Video Captioning","dataset_variant":"VidChapters-7M","rows":1,"metrics":["CIDEr"],"first_row_in_archive_order":{"model":"Vid2Seq","paper":null,"metrics":{"CIDEr":"120.5"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/chapter-llama-efficient-chaptering-in-hour-1","title":"Chapter-Llama: Efficient Chaptering in Hour-Long Videos with LLMs","date":"2025-03-31","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/revisionllm-recursive-vision-language-model","title":"ReVisionLLM: Recursive Vision-Language Model for Temporal Grounding in Hour-Long Videos","date":"2024-11-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vidchapters-7m-video-chapters-at-scale","title":"VidChapters-7M: Video Chapters at Scale","date":"2023-09-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}