{"url":"/dataset/emilia-dataset","name":"Emilia Dataset","full_name":"An Extensive, Multilingual, and Diverse Speech Dataset for Large-Scale Speech Generation","description_markdown":"Recent advancements in speech generation models have been significantly driven by the use of large-scale training data. However, producing highly spontaneous, human-like speech remains a challenge due to the scarcity of large, diverse, and spontaneous speech datasets. In response, we introduce Emilia, the first large-scale, multilingual, and diverse speech generation dataset. Emilia starts with over 101k hours of speech across six languages, covering a wide range of speaking styles to enable more natural and spontaneous speech generation. To facilitate the scale-up of Emilia, we also present Emilia-Pipe, the first open-source preprocessing pipeline designed to efficiently transform raw, in-the-wild speech data into high-quality training data with speech annotations. Experimental results demonstrate the effectiveness of both Emilia and Emilia-Pipe. Demos are available at: https://emilia-dataset.github.io/Emilia-Demo-Page/.","description_withheld":null,"homepage":"https://huggingface.co/datasets/amphion/Emilia-Dataset","introduced_date":"2024-07-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/emilia-an-extensive-multilingual-and-diverse","title":"Emilia: An Extensive, Multilingual, and Diverse Speech Dataset for Large-Scale Speech Generation","first_author":"Haorui He","url":null},"license":null,"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Zero-Shot Multi-Speaker TTS","url":"/task/zero-shot-multi-speaker-tts","datasets_with_task":"/datasets/task/zero-shot-multi-speaker-tts"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"French","url":"/datasets/language/french"},{"name":"German","url":"/datasets/language/german"},{"name":"Chinese","url":"/datasets/language/chinese"},{"name":"Japanese","url":"/datasets/language/japanese"},{"name":"Korean","url":"/datasets/language/korean"}],"variants":["Emilia Dataset"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}