{"url":"/dataset/ytseg","name":"YTSeg","full_name":null,"description_markdown":"We present YTSeg, a topically and structurally diverse benchmark for the text segmentation task based on YouTube transcriptions. The dataset comprises 19,299 videos from 393 channels, amounting to 6,533 content hours. The topics are wide-ranging, covering domains such as science, lifestyle, politics, health, economy, and technology. The videos are from various types of content formats, such as podcasts, lectures, news, corporate events \\& promotional content, and, more broadly, videos from individual content creators. We refer to the [paper](https://arxiv.org/abs/2402.17633) for further information.","description_withheld":null,"homepage":"https://huggingface.co/datasets/retkowski/ytseg","introduced_date":"2024-02-27","introduced_date_note":null,"introduced_by":{"paper":"/paper/from-text-segmentation-to-smart-chaptering-a","title":"From Text Segmentation to Smart Chaptering: A Novel Benchmark for Structuring Video Transcriptions","first_author":"Fabian Retkowski","url":null},"license":{"name":"CC BY-NC-SA 4.0","url":"https://huggingface.co/datasets/retkowski/ytseg/blob/main/LICENSE"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Text Segmentation","url":"/task/text-segmentation","datasets_with_task":"/datasets/task/text-segmentation"},{"name":"Headline Generation","url":"/task/headline-generation","datasets_with_task":"/datasets/task/headline-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["YTSeg"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/headline-generation-on-ytseg","task":"Headline Generation","dataset_variant":"YTSeg","rows":2,"metrics":["BARTScore"],"first_row_in_archive_order":{"model":"BART (no context)","paper":"/paper/from-text-segmentation-to-smart-chaptering-a","metrics":{"BARTScore":"-4.21"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/from-text-segmentation-to-smart-chaptering-a","title":"From Text Segmentation to Smart Chaptering: A Novel Benchmark for Structuring Video Transcriptions","date":"2024-02-27","rows_on_this_dataset":2,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}