{"url":"/task/speech-synthesis","name":"Speech Synthesis","slug":"speech-synthesis","description_markdown":"Speech synthesis is the task of generating speech from some other modality like text, lip movements etc. \r\n\r\nPlease note that the leaderboards here are not really comparable between studies - as they use mean opinion score as a metric and collect different samples from Amazon Mechnical Turk.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [WaveNet: A generative model for raw audio](https://deepmind.com/blog/article/wavenet-generative-model-raw-audio) )</span>","categories":[{"name":"Audio","url":"/area/audio"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1249,"papers_with_code":366,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":22,"subtasks":15,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/speech-synthesis-on-libritts","slug":"speech-synthesis-on-libritts","dataset":"LibriTTS","dataset_url":"/dataset/libritts","rows_in_archive":15,"metrics":["PESQ","M-STFT","MCD","Periodicity","V/UV F1"],"first_row_in_archive_order":{"model":"PeriodWave-Turbo-L","paper_title":"Accelerating High-Fidelity Waveform Generation via Adversarial Flow Matching Optimization","paper_url":"/paper/accelerating-high-fidelity-waveform","paper_date":"2024-08-15","arxiv_id":"2408.08019","code_links":[{"title":"sh-lee-prml/periodwave","url":"https://github.com/sh-lee-prml/periodwave"}],"syntology":null}},{"leaderboard":"/sota/speech-synthesis-on-north-american-english","slug":"speech-synthesis-on-north-american-english","dataset":"North American English","dataset_url":null,"rows_in_archive":7,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"Tacotron 2","paper_title":"Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions","paper_url":"/paper/natural-tts-synthesis-by-conditioning-wavenet","paper_date":"2017-12-16","arxiv_id":"1712.05884","code_links":[{"title":"coqui-ai/TTS","url":"https://github.com/coqui-ai/TTS"},{"title":"PaddlePaddle/PaddleSpeech","url":"https://github.com/PaddlePaddle/PaddleSpeech"},{"title":"NVIDIA/tacotron2","url":"https://github.com/NVIDIA/tacotron2"},{"title":"TensorSpeech/TensorflowTTS","url":"https://github.com/TensorSpeech/TensorflowTTS"},{"title":"Rayhane-mamah/Tacotron-2","url":"https://github.com/Rayhane-mamah/Tacotron-2"},{"title":"xcmyz/FastSpeech","url":"https://github.com/xcmyz/FastSpeech"},{"title":"Jeevesh8/Cross-Lingual-Voice-Cloning","url":"https://github.com/Jeevesh8/Cross-Lingual-Voice-Cloning"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/tree/master/research/audio/tacotron2"},{"title":"BogiHsu/Tacotron2-PyTorch","url":"https://github.com/BogiHsu/Tacotron2-PyTorch"},{"title":"dipjyoti92/SC-WaveRNN","url":"https://github.com/dipjyoti92/SC-WaveRNN"},{"title":"bfs18/tacotron2","url":"https://github.com/bfs18/tacotron2"},{"title":"keonlee9420/Comprehensive-Tacotron2","url":"https://github.com/keonlee9420/Comprehensive-Tacotron2"},{"title":"codetendolkar/tacotron-2-explained","url":"https://github.com/codetendolkar/tacotron-2-explained"},{"title":"thuhcsi/tacotron","url":"https://github.com/thuhcsi/tacotron"},{"title":"kaiidams/voice100","url":"https://github.com/kaiidams/voice100"},{"title":"kaiidams/voice100-tts","url":"https://github.com/kaiidams/voice100-tts"},{"title":"dipjyoti92/TTS-Style-Transfer","url":"https://github.com/dipjyoti92/TTS-Style-Transfer"},{"title":"dathudeptrai/TensorflowTTS","url":"https://github.com/dathudeptrai/TensorflowTTS"},{"title":"creotiv/RussianTTS-Tacotron2","url":"https://github.com/creotiv/RussianTTS-Tacotron2"},{"title":"rosinality/melgan-pytorch","url":"https://github.com/rosinality/melgan-pytorch"},{"title":"xinshengwang/Tacotron-pytorch","url":"https://github.com/xinshengwang/Tacotron-pytorch"},{"title":"izzajalandoni/tts_models","url":"https://github.com/izzajalandoni/tts_models"},{"title":"choiHkk/Transformer-TTS","url":"https://github.com/choiHkk/Transformer-TTS"},{"title":"anandaswarup/rnn-tts","url":"https://github.com/anandaswarup/rnn-tts"},{"title":"anandaswarup/TTS","url":"https://github.com/anandaswarup/TTS"},{"title":"jiean001/models_m","url":"https://github.com/jiean001/models_m/tree/main/tacotron2"},{"title":"OlaWod/my-tacotron2","url":"https://github.com/OlaWod/my-tacotron2"},{"title":"thepowerfuldeez/tacotron2","url":"https://github.com/thepowerfuldeez/tacotron2"},{"title":"alpharol/Taco_Collection","url":"https://github.com/alpharol/Taco_Collection"},{"title":"s3nh/pytorch-tacotron2","url":"https://github.com/s3nh/pytorch-tacotron2"},{"title":"martinlenglet/avtacotron2","url":"https://github.com/martinlenglet/avtacotron2"},{"title":"MindSpore-paper-code-2/code400","url":"https://github.com/MindSpore-paper-code-2/code400/tree/main/Tacotron2"},{"title":"vincenzo-scotti/tacotron2","url":"https://github.com/vincenzo-scotti/tacotron2"}],"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/speech-synthesis-on-ljspeech","slug":"speech-synthesis-on-ljspeech","dataset":"LJSpeech","dataset_url":"/dataset/ljspeech","rows_in_archive":4,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"BDDM vocoder","paper_title":"BDDM: Bilateral Denoising Diffusion Models for Fast and High-Quality Speech Synthesis","paper_url":"/paper/bddm-bilateral-denoising-diffusion-models-for-1","paper_date":"2022-03-25","arxiv_id":"2203.13508","code_links":[{"title":"tencent-ailab/bddm","url":"https://github.com/tencent-ailab/bddm"}],"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/speech-synthesis-on-mandarin-chinese","slug":"speech-synthesis-on-mandarin-chinese","dataset":"Mandarin Chinese","dataset_url":null,"rows_in_archive":3,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"WaveNet (L+F)","paper_title":"WaveNet: A Generative Model for Raw Audio","paper_url":"/paper/wavenet-a-generative-model-for-raw-audio","paper_date":"2016-09-12","arxiv_id":"1609.03499","code_links":[{"title":"ibab/tensorflow-wavenet","url":"https://github.com/ibab/tensorflow-wavenet"},{"title":"awslabs/gluon-ts","url":"https://github.com/awslabs/gluon-ts"},{"title":"karpathy/makemore","url":"https://github.com/karpathy/makemore"},{"title":"basveeling/wavenet","url":"https://github.com/basveeling/wavenet"},{"title":"vincentherrmann/pytorch-wavenet","url":"https://github.com/vincentherrmann/pytorch-wavenet"},{"title":"mindspore-ai/models","url":"https://github.com/mindspore-ai/models/tree/master/research/audio/wavenet"},{"title":"swasun/VQ-VAE-Speech","url":"https://github.com/swasun/VQ-VAE-Speech"},{"title":"WLM1ke/poptimizer","url":"https://github.com/WLM1ke/poptimizer"},{"title":"Baichenjia/Tensorflow-TCN","url":"https://github.com/Baichenjia/Tensorflow-TCN"},{"title":"MSRDL/Deep4Cast","url":"https://github.com/MSRDL/Deep4Cast"},{"title":"albarji/neurowriter","url":"https://github.com/albarji/neurowriter"},{"title":"randomrandom/deep-atrous-cnn-sentiment","url":"https://github.com/randomrandom/deep-atrous-cnn-sentiment"},{"title":"ashishpatel26/tcn-keras-Examples","url":"https://github.com/ashishpatel26/tcn-keras-Examples"},{"title":"HaiFengZeng/clari_wavenet_vocoder","url":"https://github.com/HaiFengZeng/clari_wavenet_vocoder"},{"title":"rampage644/wavenet","url":"https://github.com/rampage644/wavenet"},{"title":"ZhouYuxuanYX/Wavenet-in-Keras-for-Kaggle-Competition-Web-Traffic-Time-Series-Forecasting","url":"https://github.com/ZhouYuxuanYX/Wavenet-in-Keras-for-Kaggle-Competition-Web-Traffic-Time-Series-Forecasting"},{"title":"PeihaoChen/regnet","url":"https://github.com/PeihaoChen/regnet"},{"title":"ShichengChen/WaveNetSeparateAudio","url":"https://github.com/ShichengChen/WaveNetSeparateAudio"},{"title":"scpark20/universal-music-translation","url":"https://github.com/scpark20/universal-music-translation"},{"title":"peustr/wavenet","url":"https://github.com/peustr/wavenet"},{"title":"stdereka/liverpool-ion-switching","url":"https://github.com/stdereka/liverpool-ion-switching"},{"title":"Chasm4359/ProTS","url":"https://github.com/Chasm4359/ProTS"},{"title":"PhilippeNguyen/keras_wavenet","url":"https://github.com/PhilippeNguyen/keras_wavenet"},{"title":"TanUkkii007/wavenet","url":"https://github.com/TanUkkii007/wavenet"},{"title":"thorwhalen/slang","url":"https://github.com/thorwhalen/slang"},{"title":"otosense/slang","url":"https://github.com/otosense/slang"},{"title":"thorwhalen/sla","url":"https://github.com/thorwhalen/sla"},{"title":"zhong110020/keras-tcn","url":"https://github.com/zhong110020/keras-tcn"},{"title":"LucaHermes/lightweight-motion-forecasting","url":"https://github.com/LucaHermes/lightweight-motion-forecasting"},{"title":"kingstarcraft/speech-to-text-wavenet2","url":"https://github.com/kingstarcraft/speech-to-text-wavenet2"},{"title":"AI-Huang/WaveNet","url":"https://github.com/AI-Huang/WaveNet"},{"title":"coreyoconnor/tensorderp","url":"https://github.com/coreyoconnor/tensorderp"},{"title":"Talk2Levi/DJL","url":"https://github.com/Talk2Levi/DJL"},{"title":"r9y9/wavenet","url":"https://github.com/r9y9/wavenet"},{"title":"ShuSQ/CCI_AP_PoseLoops","url":"https://github.com/ShuSQ/CCI_AP_PoseLoops"},{"title":"yebiny/DepthOfAnaesthesia_eeg","url":"https://github.com/yebiny/DepthOfAnaesthesia_eeg"},{"title":"ShotDownDiane/tcn-master","url":"https://github.com/ShotDownDiane/tcn-master"},{"title":"isadrtdinov/wavenet","url":"https://github.com/isadrtdinov/wavenet"},{"title":"imdatsolak/wavenet","url":"https://github.com/imdatsolak/wavenet"},{"title":"DevonFulcher/CryptoPricePredictor","url":"https://github.com/DevonFulcher/CryptoPricePredictor"},{"title":"anandharaju/Basic_TCN","url":"https://github.com/anandharaju/Basic_TCN"},{"title":"ucsd-dsc-arts/dsc160-final-dsc160-final-group19","url":"https://github.com/ucsd-dsc-arts/dsc160-final-dsc160-final-group19"},{"title":"Salazar-99/Gravitational-WaveNet","url":"https://github.com/Salazar-99/Gravitational-WaveNet"},{"title":"sriharireddypusapati/speech-to-text-wavenet2","url":"https://github.com/sriharireddypusapati/speech-to-text-wavenet2"},{"title":"benmoseley/simple-wavenet","url":"https://github.com/benmoseley/simple-wavenet"},{"title":"pascalbakker/WaveNet-Implementation","url":"https://github.com/pascalbakker/WaveNet-Implementation"},{"title":"ZTianle/keras-tcn-solar","url":"https://github.com/ZTianle/keras-tcn-solar"},{"title":"Gal1eo/DT2119","url":"https://github.com/Gal1eo/DT2119"},{"title":"zhong110020/Tensorflow-TCN","url":"https://github.com/zhong110020/Tensorflow-TCN"},{"title":"adityaagrawal7/speech-to-text-wavenet","url":"https://github.com/adityaagrawal7/speech-to-text-wavenet"},{"title":"Vikas-Sony/speech-to-text","url":"https://github.com/Vikas-Sony/speech-to-text"},{"title":"outofculture/talk-like-me","url":"https://github.com/outofculture/talk-like-me"},{"title":"vicky-hnk/time-flex","url":"https://github.com/vicky-hnk/time-flex"},{"title":"HsiehKang/wavenet","url":"https://github.com/HsiehKang/wavenet/blob/main/wavenet.ipynb"},{"title":"2023-MindSpore-1/ms-code-15","url":"https://github.com/2023-MindSpore-1/ms-code-15/tree/main/wavenet"},{"title":"RamsteinWR/wavenet-master","url":"https://github.com/RamsteinWR/wavenet-master"},{"title":"freedombenLiu/speech-to-text-wavenet","url":"https://github.com/freedombenLiu/speech-to-text-wavenet"},{"title":"liguigui/speech-to-text-wavenet","url":"https://github.com/liguigui/speech-to-text-wavenet"},{"title":"glakshay/Generating-audio-DL","url":"https://github.com/glakshay/Generating-audio-DL"},{"title":"Shivendra-psc/speechbot","url":"https://github.com/Shivendra-psc/speechbot"},{"title":"pbrandl/aNN_Audio","url":"https://github.com/pbrandl/aNN_Audio"},{"title":"zll1996/TCN","url":"https://github.com/zll1996/TCN"}],"syntology":{"n":103,"n_ran":41,"n_unverified":62,"n_pointer_only":25}}},{"leaderboard":"/sota/speech-synthesis-on-blizzard-challenge-2013","slug":"speech-synthesis-on-blizzard-challenge-2013","dataset":"Blizzard Challenge 2013","dataset_url":"/dataset/blizzard-challenge-2013","rows_in_archive":2,"metrics":["NLL"],"first_row_in_archive_order":{"model":"SampleRNN (3-tier)","paper_title":"SampleRNN: An Unconditional End-to-End Neural Audio Generation Model","paper_url":"/paper/samplernn-an-unconditional-end-to-end-neural","paper_date":"2016-12-22","arxiv_id":"1612.07837","code_links":[{"title":"soroushmehr/sampleRNN_ICLR2017","url":"https://github.com/soroushmehr/sampleRNN_ICLR2017"},{"title":"deepsound-project/samplernn-pytorch","url":"https://github.com/deepsound-project/samplernn-pytorch"},{"title":"dada-bots/dadabots_sampleRNN","url":"https://github.com/dada-bots/dadabots_sampleRNN"},{"title":"cchinchristopherj/Concert-of-Whales","url":"https://github.com/cchinchristopherj/Concert-of-Whales"}],"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/ljspeech","name":"LJSpeech","full_name":"The LJ Speech Dataset","num_papers_in_archive":323},{"url":"/dataset/libritts","name":"LibriTTS","full_name":"LibriTTS","num_papers_in_archive":257},{"url":"/dataset/thchs-30","name":"THCHS-30","full_name":"","num_papers_in_archive":34},{"url":"/dataset/css10","name":"CSS10","full_name":"","num_papers_in_archive":22},{"url":"/dataset/promptspeech","name":"PromptSpeech","full_name":"","num_papers_in_archive":14},{"url":"/dataset/somos","name":"SOMOS","full_name":"The Samsung Open MOS Dataset for the Evaluation of Neural Text-to-Speech Synthesis","num_papers_in_archive":11},{"url":"/dataset/jsut-corpus","name":"JSUT Corpus","full_name":"","num_papers_in_archive":8},{"url":"/dataset/gumar-corpus","name":"Gumar Corpus","full_name":"","num_papers_in_archive":7},{"url":"/dataset/blizzard-challenge-2013","name":"Blizzard Challenge 2013","full_name":"Blizzard Challenge 2013 - English language tasks","num_papers_in_archive":5},{"url":"/dataset/hui","name":"HUI speech corpus","full_name":"Hof University iisys speech dataset","num_papers_in_archive":3},{"url":"/dataset/tal-corpus","name":"TaL Corpus","full_name":"The Tongue and Lips Corpus","num_papers_in_archive":3},{"url":"/dataset/jit-dataset","name":"JIT Dataset","full_name":"Jejueo Interview Transcripts","num_papers_in_archive":2},{"url":"/dataset/jvs-music","name":"JVS-MuSiC","full_name":"","num_papers_in_archive":2},{"url":"/dataset/ruslan","name":"RUSLAN","full_name":"","num_papers_in_archive":2},{"url":"/dataset/tilde-model-corpus","name":"Tilde MODEL Corpus","full_name":"Tilde Multilingual Open Data for European Languages","num_papers_in_archive":2},{"url":"/dataset/tts-portuguese-corpus","name":"TTS-Portuguese Corpus","full_name":"","num_papers_in_archive":2},{"url":"/dataset/vocbench","name":"VocBench","full_name":"","num_papers_in_archive":2},{"url":"/dataset/gneutralspeech-male","name":"GneutralSpeech Male","full_name":"","num_papers_in_archive":1},{"url":"/dataset/jss-dataset","name":"JSS Dataset","full_name":"Jejueo Single Speaker Speech","num_papers_in_archive":1},{"url":"/dataset/silent-speech-emg","name":"Silent Speech EMG","full_name":"","num_papers_in_archive":1},{"url":"/dataset/ttsds-synthetic-speech","name":"TTSDS Synthetic Speech","full_name":"","num_papers_in_archive":1},{"url":"/dataset/united-syn-med","name":"United-Syn-Med","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/emotional-speech-synthesis","name":"Emotional Speech Synthesis"},{"url":"/task/expressive-speech-synthesis","name":"Expressive Speech Synthesis"},{"url":"/task/speech-synthesis-assamese","name":"Speech Synthesis - Assamese"},{"url":"/task/speech-synthesis-bengali","name":"Speech Synthesis - Bengali"},{"url":"/task/speech-synthesis-bodo","name":"Speech Synthesis - Bodo"},{"url":"/task/speech-synthesis-gujarati","name":"Speech Synthesis - Gujarati"},{"url":"/task/speech-synthesis-hindi","name":"Speech Synthesis - Hindi"},{"url":"/task/speech-synthesis-kannada","name":"Speech Synthesis - Kannada"},{"url":"/task/speech-synthesis-malayalam","name":"Speech Synthesis - Malayalam"},{"url":"/task/speech-synthesis-manipuri","name":"Speech Synthesis - Manipuri"},{"url":"/task/speech-synthesis-marathi","name":"Speech Synthesis - Marathi"},{"url":"/task/speech-synthesis-rajasthani","name":"Speech Synthesis - Rajasthani"},{"url":"/task/speech-synthesis-tamil","name":"Speech Synthesis - Tamil"},{"url":"/task/speech-synthesis-telugu","name":"Speech Synthesis - Telugu"},{"url":"/task/text-to-speech-translation","name":"text-to-speech translation"}],"parent_tasks":[{"url":"/task/accented-speech-recognition","name":"Accented Speech Recognition"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":366,"tagged_in_all":1249,"items":[{"url":"/paper/wavenet-a-generative-model-for-raw-audio","title":"WaveNet: A Generative Model for Raw Audio","date":"2016-09-12","arxiv_id":"1609.03499","repositories_listed":62,"syntology":{"n":103,"n_ran":41,"n_unverified":62,"n_pointer_only":25}},{"url":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","repositories_listed":37,"syntology":{"n":119,"n_ran":73,"n_unverified":46,"n_pointer_only":33}},{"url":"/paper/natural-tts-synthesis-by-conditioning-wavenet","title":"Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions","date":"2017-12-16","arxiv_id":"1712.05884","repositories_listed":33,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/tacotron-towards-end-to-end-speech-synthesis","title":"Tacotron: Towards End-to-End Speech Synthesis","date":"2017-03-29","arxiv_id":"1703.10135","repositories_listed":30,"syntology":{"n":25,"n_ran":7,"n_unverified":18,"n_pointer_only":6}},{"url":"/paper/fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","repositories_listed":22,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":3}},{"url":"/paper/melgan-generative-adversarial-networks-for","title":"MelGAN: Generative Adversarial Networks for Conditional Waveform Synthesis","date":"2019-10-08","arxiv_id":"1910.06711","repositories_listed":21,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","arxiv_id":"1910.11480","repositories_listed":12,"syntology":{"n":20,"n_ran":0,"n_unverified":20,"n_pointer_only":1}},{"url":"/paper/hifi-gan-generative-adversarial-networks-for","title":"HiFi-GAN: Generative Adversarial Networks for Efficient and High Fidelity Speech Synthesis","date":"2020-10-12","arxiv_id":"2010.05646","repositories_listed":11,"syntology":{"n":25,"n_ran":17,"n_unverified":8,"n_pointer_only":3}},{"url":"/paper/diffwave-a-versatile-diffusion-model-for","title":"DiffWave: A Versatile Diffusion Model for Audio Synthesis","date":"2020-09-21","arxiv_id":"2009.09761","repositories_listed":11,"syntology":{"n":33,"n_ran":20,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/transfer-learning-from-speaker-verification","title":"Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis","date":"2018-06-12","arxiv_id":"1806.04558","repositories_listed":11,"syntology":null},{"url":"/paper/style-tokens-unsupervised-style-modeling","title":"Style Tokens: Unsupervised Style Modeling, Control and Transfer in End-to-End Speech Synthesis","date":"2018-03-23","arxiv_id":"1803.09017","repositories_listed":11,"syntology":{"n":21,"n_ran":6,"n_unverified":15,"n_pointer_only":7}},{"url":"/paper/univnet-a-neural-vocoder-with-multi","title":"UnivNet: A Neural Vocoder with Multi-Resolution Spectrogram Discriminators for High-Fidelity Waveform Generation","date":"2021-06-15","arxiv_id":"2106.07889","repositories_listed":9,"syntology":{"n":12,"n_ran":8,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/wavegrad-estimating-gradients-for-waveform","title":"WaveGrad: Estimating Gradients for Waveform Generation","date":"2020-09-02","arxiv_id":"2009.00713","repositories_listed":7,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/deep-voice-3-scaling-text-to-speech-with","title":"Deep Voice 3: Scaling Text-to-Speech with Convolutional Sequence Learning","date":"2017-10-20","arxiv_id":"1710.07654","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","arxiv_id":"2110.07205","repositories_listed":6,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","arxiv_id":"2105.06337","repositories_listed":6,"syntology":{"n":18,"n_ran":9,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/durian-duration-informed-attention-network","title":"DurIAN: Duration Informed Attention Network For Multimodal Synthesis","date":"2019-09-04","arxiv_id":"1909.01700","repositories_listed":6,"syntology":null},{"url":"/paper/neural-speech-synthesis-with-transformer","title":"Neural Speech Synthesis with Transformer Network","date":"2018-09-19","arxiv_id":"1809.08895","repositories_listed":6,"syntology":null},{"url":"/paper/neural-autoregressive-flows","title":"Neural Autoregressive Flows","date":"2018-04-03","arxiv_id":"1804.00779","repositories_listed":6,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":2}},{"url":"/paper/bigvgan-a-universal-neural-vocoder-with-large","title":"BigVGAN: A Universal Neural Vocoder with Large-Scale Training","date":"2022-06-09","arxiv_id":"2206.04658","repositories_listed":5,"syntology":{"n":17,"n_ran":8,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/melnet-a-generative-model-for-audio-in-the","title":"MelNet: A Generative Model for Audio in the Frequency Domain","date":"2019-06-04","arxiv_id":"1906.01083","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/exploring-transfer-learning-for-low-resource","title":"Exploring Transfer Learning for Low Resource Emotional TTS","date":"2019-01-14","arxiv_id":"1901.04276","repositories_listed":5,"syntology":null},{"url":"/paper/clarinet-parallel-wave-generation-in-end-to","title":"ClariNet: Parallel Wave Generation in End-to-End Text-to-Speech","date":"2018-07-19","arxiv_id":"1807.07281","repositories_listed":5,"syntology":{"n":13,"n_ran":1,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/statistical-parametric-speech-synthesis-1","title":"Statistical Parametric Speech Synthesis Incorporating Generative Adversarial Networks","date":"2017-09-23","arxiv_id":"1709.08041","repositories_listed":5,"syntology":null},{"url":"/paper/speech-slytherin-examining-the-performance","title":"Speech Slytherin: Examining the Performance and Efficiency of Mamba for Speech Separation, Recognition, and Synthesis","date":"2024-07-13","arxiv_id":"2407.09732","repositories_listed":4,"syntology":null},{"url":"/paper/vocos-closing-the-gap-between-time-domain-and","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","date":"2023-06-01","arxiv_id":"2306.00814","repositories_listed":4,"syntology":{"n":16,"n_ran":5,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/scaling-speech-technology-to-1000-languages-1","title":"Scaling Speech Technology to 1,000+ Languages","date":"2023-05-22","arxiv_id":"2305.13516","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/prodiff-progressive-fast-diffusion-model-for","title":"ProDiff: Progressive Fast Diffusion Model For High-Quality Text-to-Speech","date":"2022-07-13","arxiv_id":"2207.06389","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}}],"syntology_records":24,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}