{"url":"/method/glow-tts","slug":"glow-tts","name":"Glow-TTS","full_name":"Glow-TTS","full_name_withheld":false,"description_markdown":"**Glow-TTS** is a flow-based generative model for parallel TTS that does not require any external aligner. By combining the properties of flows and dynamic programming, the proposed model searches for the most probable monotonic alignment between text and the latent representation of speech.  The model is directly trained to maximize the log-likelihood of speech with the alignment. Enforcing hard monotonic alignments helps enable robust TTS, which generalizes to long utterances, and employing flows enables fast, diverse, and controllable speech synthesis.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","paper":"/paper/glow-tts-a-generative-flow-for-text-to-speech","first_author":"Jaehyeon Kim","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/glow-tts-a-generative-flow-for-text-to-speech"},"source":{"url":"https://arxiv.org/abs/2005.11129v1","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Audio","area_id":"audio","collection":"Text-to-Speech Models","url":"/methods/category/text-to-speech-models","pwc_aliases":[]}],"n_papers_tagged":7,"archive_num_papers":7,"papers_newest_first":[{"paper":"/paper/super-monotonic-alignment-search","title":"Super Monotonic Alignment Search","date":"2024-09-12","arxiv_id":"2409.07704","n_code_links":1,"syntology":null},{"paper":"/paper/stochastic-pitch-prediction-improves-the","title":"Stochastic Pitch Prediction Improves the Diversity and Naturalness of Speech in Glow-TTS","date":"2023-05-28","arxiv_id":"2305.17724","n_code_links":1,"syntology":null},{"paper":null,"title":"ClArTTS: An Open-Source Classical Arabic Text-to-Speech Corpus","date":"2023-02-28","arxiv_id":"2303.00069","n_code_links":0,"syntology":null},{"paper":null,"title":"GlowVC: Mel-spectrogram space disentangling model for language-independent text-free voice conversion","date":"2022-07-04","arxiv_id":"2207.01454","n_code_links":0,"syntology":null},{"paper":"/paper/portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","n_code_links":4,"syntology":{"ran":12,"of":12,"unverified":0,"pointer_only":1}},{"paper":"/paper/bidirectional-variational-inference-for-non","title":"Bidirectional Variational Inference for Non-Autoregressive Text-to-Speech","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/glow-tts-a-generative-flow-for-text-to-speech","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","date":"2020-05-22","arxiv_id":"2005.11129","n_code_links":6,"syntology":{"ran":8,"of":14,"unverified":6,"pointer_only":0}}],"papers_shown":7,"tasks":[{"task":"/task/text-to-speech","name":"Text to Speech","papers":5},{"task":"/task/text-to-speech-1","name":"text-to-speech","papers":5},{"task":"/task/text-to-speech-synthesis","name":"Text-To-Speech Synthesis","papers":2},{"task":null,"name":"CPU","papers":1},{"task":"/task/diversity","name":"Diversity","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/speech-synthesis","name":"Speech Synthesis","papers":1},{"task":"/task/variational-inference","name":"Variational Inference","papers":1},{"task":"/task/high","name":"Vocal Bursts Intensity Prediction","papers":1},{"task":"/task/voice-conversion","name":"Voice Conversion","papers":1},{"task":"/task/word-alignment","name":"Word Alignment","papers":1},{"task":"/task/zero-shot-multi-speaker-tts","name":"Zero-Shot Multi-Speaker TTS","papers":1}],"tasks_shown":12,"n_tasks":12,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":2},{"year":"2022","papers":1},{"year":"2023","papers":2},{"year":"2024","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/glow-tts"},"syntology_read_at":"2026-09-25T09:33:49+00:00"}