{"url":"/dataset/spotify-podcast","name":"Spotify Podcast","full_name":null,"description_markdown":"A set of approximately 100K podcast episodes comprised of raw audio files along with accompanying ASR transcripts. This represents over 47,000 hours of transcribed audio, and is an order of magnitude larger than previous speech-to-text corpora. \r\n\r\nSource: [The Spotify Podcast Dataset](/paper/the-spotify-podcasts-dataset)","description_withheld":null,"homepage":"https://podcastsdataset.byspotify.com/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/the-spotify-podcasts-dataset","title":"The Spotify Podcast Dataset","first_author":"Ann Clifton","url":null},"license":null,"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Recommendation Systems","url":"/task/recommendation-systems","datasets_with_task":"/datasets/task/recommendation-systems"},{"name":"Abstractive Text Summarization","url":"/task/abstractive-text-summarization","datasets_with_task":"/datasets/task/abstractive-text-summarization"}],"languages":[],"variants":["Spotify Podcast"],"data_loaders":[],"num_papers_in_archive":6,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}