{"url":"/task/text-to-speech-synthesis","name":"Text-To-Speech Synthesis","slug":"text-to-speech-synthesis","description_markdown":"**Text-To-Speech Synthesis** is a machine learning task that involves converting written text into spoken words. The goal is to generate synthetic speech that sounds natural and resembles human speech as closely as possible.","categories":[{"name":"Audio","url":"/area/audio"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":332,"papers_with_code":104,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":18,"subtasks":2,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/text-to-speech-synthesis-on-ljspeech","slug":"text-to-speech-synthesis-on-ljspeech","dataset":"LJSpeech","dataset_url":"/dataset/ljspeech","rows_in_archive":16,"metrics":["Audio Quality MOS","Pleasantness MOS","Word Error Rate (WER)","MOS","WER (%)"],"first_row_in_archive_order":{"model":"NaturalSpeech","paper_title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","paper_url":"/paper/naturalspeech-end-to-end-text-to-speech","paper_date":"2022-05-09","arxiv_id":"2205.04421","code_links":[{"title":"microsoft/NeuralSpeech","url":"https://github.com/microsoft/NeuralSpeech"},{"title":"daniilrobnikov/vits2","url":"https://github.com/daniilrobnikov/vits2"},{"title":"heatz123/naturalspeech","url":"https://github.com/heatz123/naturalspeech"}],"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}}},{"leaderboard":"/sota/text-to-speech-synthesis-on-20000-utterances","slug":"text-to-speech-synthesis-on-20000-utterances","dataset":"20000 utterances","dataset_url":"/dataset/20000-utterances","rows_in_archive":1,"metrics":["10-keyword Speech Commands dataset"],"first_row_in_archive_order":{"model":"Mia","paper_title":"MIA-Prognosis: A Deep Learning Framework to Predict Therapy Response","paper_url":"/paper/mia-prognosis-a-deep-learning-framework-to","paper_date":"2020-10-08","arxiv_id":"2010.04062","code_links":[{"title":"M3DV/SimTA","url":"https://github.com/M3DV/SimTA"}],"syntology":null}},{"leaderboard":"/sota/text-to-speech-synthesis-on-cmudict-07b","slug":"text-to-speech-synthesis-on-cmudict-07b","dataset":"CMUDict 0.7b","dataset_url":null,"rows_in_archive":1,"metrics":["Phoneme Error Rate","Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"Token-Level Ensemble Distillation","paper_title":"Token-Level Ensemble Distillation for Grapheme-to-Phoneme Conversion","paper_url":"/paper/token-level-ensemble-distillation-for","paper_date":"2019-04-06","arxiv_id":"1904.03446","code_links":[],"syntology":null}},{"leaderboard":"/sota/text-to-speech-synthesis-on-hui","slug":"text-to-speech-synthesis-on-hui","dataset":"HUI speech corpus","dataset_url":"/dataset/hui","rows_in_archive":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"Tacotron 2","paper_title":"Neural Speech Synthesis in German","paper_url":"/paper/neural-speech-synthesis-in-german","paper_date":"2021-10-03","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/text-to-speech-synthesis-on-thorsten-voice-21","slug":"text-to-speech-synthesis-on-thorsten-voice-21","dataset":"Thorsten voice 21.02 neutral","dataset_url":"/dataset/thorsten-voice-21-02-neutral","rows_in_archive":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"Tacotron 2","paper_title":"Neural Speech Synthesis in German","paper_url":"/paper/neural-speech-synthesis-in-german","paper_date":"2021-10-03","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/text-to-speech-synthesis-on-trinity-speech","slug":"text-to-speech-synthesis-on-trinity-speech","dataset":"Trinity Speech-Gesture Dataset","dataset_url":"/dataset/https-trinityspeechgesture-scss-tcd-ie","rows_in_archive":1,"metrics":["MOS"],"first_row_in_archive_order":{"model":"Match-TTSG","paper_title":"Unified speech and gesture synthesis using flow matching","paper_url":"/paper/unified-speech-and-gesture-synthesis-using","paper_date":"2023-10-08","arxiv_id":"2310.05181","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/ljspeech","name":"LJSpeech","full_name":"The LJ Speech Dataset","num_papers_in_archive":323},{"url":"/dataset/libritts","name":"LibriTTS","full_name":"LibriTTS","num_papers_in_archive":257},{"url":"/dataset/aishell-3","name":"AISHELL-3","full_name":"","num_papers_in_archive":38},{"url":"/dataset/cvss","name":"CVSS","full_name":"","num_papers_in_archive":26},{"url":"/dataset/20000-utterances","name":"20000 utterances","full_name":"20000 utterances","num_papers_in_archive":15},{"url":"/dataset/somos","name":"SOMOS","full_name":"The Samsung Open MOS Dataset for the Evaluation of Neural Text-to-Speech Synthesis","num_papers_in_archive":11},{"url":"/dataset/gumar-corpus","name":"Gumar Corpus","full_name":"","num_papers_in_archive":7},{"url":"/dataset/kazakhtts","name":"KazakhTTS","full_name":"","num_papers_in_archive":4},{"url":"/dataset/speechinstruct","name":"SpeechInstruct","full_name":"","num_papers_in_archive":4},{"url":"/dataset/emovie","name":"EMOVIE","full_name":"","num_papers_in_archive":3},{"url":"/dataset/hui","name":"HUI speech corpus","full_name":"Hof University iisys speech dataset","num_papers_in_archive":3},{"url":"/dataset/openslr","name":"OpenSLR","full_name":"Open Speech and Language Resources","num_papers_in_archive":2},{"url":"/dataset/ryanspeech","name":"RyanSpeech","full_name":"","num_papers_in_archive":2},{"url":"/dataset/https-trinityspeechgesture-scss-tcd-ie","name":"Trinity Speech-Gesture Dataset","full_name":"","num_papers_in_archive":2},{"url":"/dataset/gneutralspeech-female","name":"GneutralSpeech Female","full_name":"","num_papers_in_archive":1},{"url":"/dataset/gneutralspeech-male","name":"GneutralSpeech Male","full_name":"","num_papers_in_archive":1},{"url":"/dataset/imasc","name":"IMaSC","full_name":"ICFOSS Malayalam Speech Corpus","num_papers_in_archive":1},{"url":"/dataset/thorsten-voice-21-02-neutral","name":"Thorsten voice 21.02 neutral","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/prosody-prediction","name":"Prosody Prediction"},{"url":"/task/zero-shot-multi-speaker-tts","name":"Zero-Shot Multi-Speaker TTS"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":104,"tagged_in_all":332,"items":[{"url":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","repositories_listed":37,"syntology":{"n":119,"n_ran":73,"n_unverified":46,"n_pointer_only":33}},{"url":"/paper/tacotron-towards-end-to-end-speech-synthesis","title":"Tacotron: Towards End-to-End Speech Synthesis","date":"2017-03-29","arxiv_id":"1703.10135","repositories_listed":30,"syntology":{"n":25,"n_ran":7,"n_unverified":18,"n_pointer_only":6}},{"url":"/paper/fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","repositories_listed":22,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":3}},{"url":"/paper/efficiently-trainable-text-to-speech-system","title":"Efficiently Trainable Text-to-Speech System Based on Deep Convolutional Networks with Guided Attention","date":"2017-10-24","arxiv_id":"1710.08969","repositories_listed":22,"syntology":{"n":28,"n_ran":1,"n_unverified":27,"n_pointer_only":1}},{"url":"/paper/efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","arxiv_id":"1910.11480","repositories_listed":12,"syntology":{"n":20,"n_ran":0,"n_unverified":20,"n_pointer_only":1}},{"url":"/paper/fastspeech-fastrobustand-controllable-text-to","title":"FastSpeech: Fast,Robustand Controllable Text-to-Speech","date":"2019-05-22","arxiv_id":null,"repositories_listed":11,"syntology":null},{"url":"/paper/transfer-learning-from-speaker-verification","title":"Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis","date":"2018-06-12","arxiv_id":"1806.04558","repositories_listed":11,"syntology":null},{"url":"/paper/style-tokens-unsupervised-style-modeling","title":"Style Tokens: Unsupervised Style Modeling, Control and Transfer in End-to-End Speech Synthesis","date":"2018-03-23","arxiv_id":"1803.09017","repositories_listed":11,"syntology":{"n":21,"n_ran":6,"n_unverified":15,"n_pointer_only":7}},{"url":"/paper/diffsinger-diffusion-acoustic-model-for","title":"DiffSinger: Singing Voice Synthesis via Shallow Diffusion Mechanism","date":"2021-05-06","arxiv_id":"2105.02446","repositories_listed":10,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/wavegrad-estimating-gradients-for-waveform","title":"WaveGrad: Estimating Gradients for Waveform Generation","date":"2020-09-02","arxiv_id":"2009.00713","repositories_listed":7,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","arxiv_id":"2105.06337","repositories_listed":6,"syntology":{"n":18,"n_ran":9,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/glow-tts-a-generative-flow-for-text-to-speech","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","date":"2020-05-22","arxiv_id":"2005.11129","repositories_listed":6,"syntology":{"n":14,"n_ran":3,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/neural-speech-synthesis-with-transformer","title":"Neural Speech Synthesis with Transformer Network","date":"2018-09-19","arxiv_id":"1809.08895","repositories_listed":6,"syntology":null},{"url":"/paper/melnet-a-generative-model-for-audio-in-the","title":"MelNet: A Generative Model for Audio in the Frequency Domain","date":"2019-06-04","arxiv_id":"1906.01083","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/exploring-transfer-learning-for-low-resource","title":"Exploring Transfer Learning for Low Resource Emotional TTS","date":"2019-01-14","arxiv_id":"1901.04276","repositories_listed":5,"syntology":null},{"url":"/paper/prodiff-progressive-fast-diffusion-model-for","title":"ProDiff: Progressive Fast Diffusion Model For High-Quality Text-to-Speech","date":"2022-07-13","arxiv_id":"2207.06389","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","repositories_listed":4,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/tools-and-resources-for-romanian-text-to","title":"Tools and resources for Romanian text-to-speech and speech-to-text applications","date":"2018-02-15","arxiv_id":"1802.05583","repositories_listed":4,"syntology":null},{"url":"/paper/enhancing-suno-s-bark-text-to-speech-model","title":"Enhancing Suno's Bark Text-to-Speech Model: Addressing Limitations Through Meta's Encodec and Pre-Trained Hubert","date":"2023-04-18","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/preparing-an-endangered-language-for-the","title":"Preparing an Endangered Language for the Digital Age: The Case of Judeo-Spanish","date":"2022-05-31","arxiv_id":"2205.15599","repositories_listed":3,"syntology":null},{"url":"/paper/naturalspeech-end-to-end-text-to-speech","title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","date":"2022-05-09","arxiv_id":"2205.04421","repositories_listed":3,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/yourtts-towards-zero-shot-multi-speaker-tts","title":"YourTTS: Towards Zero-Shot Multi-Speaker TTS and Zero-Shot Voice Conversion for everyone","date":"2021-12-04","arxiv_id":"2112.02418","repositories_listed":3,"syntology":null},{"url":"/paper/wavegrad-2-iterative-refinement-for-text-to","title":"WaveGrad 2: Iterative Refinement for Text-to-Speech Synthesis","date":"2021-06-17","arxiv_id":"2106.09660","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/ryanspeech-a-corpus-for-conversational-text","title":"RyanSpeech: A Corpus for Conversational Text-to-Speech Synthesis","date":"2021-06-15","arxiv_id":"2106.08468","repositories_listed":3,"syntology":null},{"url":"/paper/flowtron-an-autoregressive-flow-based","title":"Flowtron: an Autoregressive Flow-based Generative Network for Text-to-Speech Synthesis","date":"2020-05-12","arxiv_id":"2005.05957","repositories_listed":3,"syntology":{"n":19,"n_ran":4,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/back-transcription-as-a-method-for-evaluating","title":"Back Transcription as a Method for Evaluating Robustness of Natural Language Understanding Models to Speech Recognition Errors","date":"2023-10-25","arxiv_id":"2310.16609","repositories_listed":2,"syntology":null},{"url":"/paper/lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/towards-building-text-to-speech-systems-for","title":"Towards Building Text-To-Speech Systems for the Next Billion Users","date":"2022-11-17","arxiv_id":"2211.09536","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}],"syntology_records":20,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}