{"url":"/dataset/esd","name":"ESD","full_name":"Emotional Speech Database","description_markdown":"**ESD** is an Emotional Speech Database for voice conversion research. The ESD database consists of 350 parallel utterances spoken by 10 native English and 10 native Chinese speakers and covers 5 emotion categories (neutral, happy, angry, sad and surprise). More than 29 hours of speech data were recorded in a controlled acoustic environment. The database is suitable for multi-speaker and cross-lingual emotional voice conversion studies.","description_withheld":null,"homepage":"https://hltsingapore.github.io/ESD/","introduced_date":"2021-05-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/emotional-voice-conversion-theory-databases","title":"Emotional Voice Conversion: Theory, Databases and ESD","first_author":"Kun Zhou","url":null},"license":{"name":"Custom (research-only)","url":null},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Voice Conversion","url":"/task/voice-conversion","datasets_with_task":"/datasets/task/voice-conversion"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["ESD"],"data_loaders":[{"repo":"https://github.com/huggingface/parler-tts","url":"https://github.com/huggingface/parler-tts","frameworks":["tf","pytorch"]}],"num_papers_in_archive":63,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}