{"url":"/task/text-to-speech-1","name":"text-to-speech","slug":"text-to-speech-1","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1413,"papers_with_code":395,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/mnist","name":"MNIST","full_name":"","num_papers_in_archive":7651}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":395,"tagged_in_all":1413,"items":[{"url":"/paper/wavenet-a-generative-model-for-raw-audio","title":"WaveNet: A Generative Model for Raw Audio","date":"2016-09-12","arxiv_id":"1609.03499","repositories_listed":62,"syntology":{"n":103,"n_ran":41,"n_unverified":62,"n_pointer_only":25}},{"url":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","repositories_listed":37,"syntology":{"n":119,"n_ran":73,"n_unverified":46,"n_pointer_only":33}},{"url":"/paper/tacotron-towards-end-to-end-speech-synthesis","title":"Tacotron: Towards End-to-End Speech Synthesis","date":"2017-03-29","arxiv_id":"1703.10135","repositories_listed":30,"syntology":{"n":25,"n_ran":7,"n_unverified":18,"n_pointer_only":6}},{"url":"/paper/fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","repositories_listed":22,"syntology":{"n":11,"n_ran":3,"n_unverified":8,"n_pointer_only":3}},{"url":"/paper/efficiently-trainable-text-to-speech-system","title":"Efficiently Trainable Text-to-Speech System Based on Deep Convolutional Networks with Guided Attention","date":"2017-10-24","arxiv_id":"1710.08969","repositories_listed":22,"syntology":{"n":28,"n_ran":1,"n_unverified":27,"n_pointer_only":1}},{"url":"/paper/efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","arxiv_id":"1910.11480","repositories_listed":12,"syntology":{"n":20,"n_ran":0,"n_unverified":20,"n_pointer_only":1}},{"url":"/paper/fastspeech-fastrobustand-controllable-text-to","title":"FastSpeech: Fast,Robustand Controllable Text-to-Speech","date":"2019-05-22","arxiv_id":null,"repositories_listed":11,"syntology":null},{"url":"/paper/transfer-learning-from-speaker-verification","title":"Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis","date":"2018-06-12","arxiv_id":"1806.04558","repositories_listed":11,"syntology":null},{"url":"/paper/diffsinger-diffusion-acoustic-model-for","title":"DiffSinger: Singing Voice Synthesis via Shallow Diffusion Mechanism","date":"2021-05-06","arxiv_id":"2105.02446","repositories_listed":10,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/univnet-a-neural-vocoder-with-multi","title":"UnivNet: A Neural Vocoder with Multi-Resolution Spectrogram Discriminators for High-Fidelity Waveform Generation","date":"2021-06-15","arxiv_id":"2106.07889","repositories_listed":9,"syntology":{"n":12,"n_ran":8,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/robust-universal-neural-vocoding","title":"Robust universal neural vocoding","date":"2018-11-15","arxiv_id":"1811.06292","repositories_listed":8,"syntology":null},{"url":"/paper/neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/deep-voice-3-scaling-text-to-speech-with","title":"Deep Voice 3: Scaling Text-to-Speech with Convolutional Sequence Learning","date":"2017-10-20","arxiv_id":"1710.07654","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/sequence-transduction-with-recurrent-neural","title":"Sequence Transduction with Recurrent Neural Networks","date":"2012-11-14","arxiv_id":"1211.3711","repositories_listed":7,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/soundstream-an-end-to-end-neural-audio-codec","title":"SoundStream: An End-to-End Neural Audio Codec","date":"2021-07-07","arxiv_id":"2107.03312","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","arxiv_id":"2105.06337","repositories_listed":6,"syntology":{"n":18,"n_ran":9,"n_unverified":9,"n_pointer_only":1}},{"url":"/paper/non-attentive-tacotron-robust-and-1","title":"Non-Attentive Tacotron: Robust and Controllable Neural TTS Synthesis Including Unsupervised Duration Modeling","date":"2020-10-08","arxiv_id":"2010.04301","repositories_listed":6,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/fastpitch-parallel-text-to-speech-with-pitch","title":"FastPitch: Parallel Text-to-speech with Pitch Prediction","date":"2020-06-11","arxiv_id":"2006.06873","repositories_listed":6,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/defense-for-black-box-attacks-on-anti","title":"Defense for Black-box Attacks on Anti-spoofing Models by Self-Supervised Learning","date":"2020-06-05","arxiv_id":"2006.03214","repositories_listed":6,"syntology":null},{"url":"/paper/glow-tts-a-generative-flow-for-text-to-speech","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","date":"2020-05-22","arxiv_id":"2005.11129","repositories_listed":6,"syntology":{"n":14,"n_ran":3,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/neural-speech-synthesis-with-transformer","title":"Neural Speech Synthesis with Transformer Network","date":"2018-09-19","arxiv_id":"1809.08895","repositories_listed":6,"syntology":null},{"url":"/paper/melnet-a-generative-model-for-audio-in-the","title":"MelNet: A Generative Model for Audio in the Frequency Domain","date":"2019-06-04","arxiv_id":"1906.01083","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/clarinet-parallel-wave-generation-in-end-to","title":"ClariNet: Parallel Wave Generation in End-to-End Text-to-Speech","date":"2018-07-19","arxiv_id":"1807.07281","repositories_listed":5,"syntology":{"n":13,"n_ran":1,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/statistical-parametric-speech-synthesis-1","title":"Statistical Parametric Speech Synthesis Incorporating Generative Adversarial Networks","date":"2017-09-23","arxiv_id":"1709.08041","repositories_listed":5,"syntology":null},{"url":"/paper/seamlessm4t-massively-multilingual-multimodal","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","date":"2023-08-22","arxiv_id":"2308.11596","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/prodiff-progressive-fast-diffusion-model-for","title":"ProDiff: Progressive Fast Diffusion Model For High-Quality Text-to-Speech","date":"2022-07-13","arxiv_id":"2207.06389","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","repositories_listed":4,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/fairseq-s-2-a-scalable-and-integrable-speech","title":"fairseq S^2: A Scalable and Integrable Speech Synthesis Toolkit","date":"2021-09-14","arxiv_id":"2109.06912","repositories_listed":4,"syntology":null},{"url":"/paper/lightspeech-lightweight-and-fast-text-to","title":"LightSpeech: Lightweight and Fast Text to Speech with Neural Architecture Search","date":"2021-02-08","arxiv_id":"2102.04040","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":23,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}