{"url":"/dataset/sep-28k","name":"SEP-28k","full_name":"Stuttering Events in Podcasts","description_markdown":"Stuttering Events in Podcasts (SEP-28k) is a dataset containing over 28k clips labeled with five event types including blocks, prolongations, sound repetitions, word repetitions, and interjections. Audio comes from public podcasts largely consisting of people who stutter interviewing other people who stutter. \r\n\r\nSource: [Lea et al.](https://arxiv.org/pdf/2102.12394.pdf)\r\n\r\nImage source: [Lea et al.](https://arxiv.org/pdf/2102.12394.pdf)","description_withheld":null,"homepage":"https://arxiv.org/pdf/2102.12394.pdf","introduced_date":"2021-02-24","introduced_date_note":null,"introduced_by":{"paper":"/paper/sep-28k-a-dataset-for-stuttering-event","title":"SEP-28k: A Dataset for Stuttering Event Detection From Podcasts With People Who Stutter","first_author":"Colin Lea","url":null},"license":null,"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["SEP-28k"],"data_loaders":[],"num_papers_in_archive":21,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}