{"url":"/task/text-to-speech","name":"Text to Speech","slug":"text-to-speech","description_markdown":"import gTTS import os def text_to_speech_kurdish(text, output_file=\"output.mp3\"): # گۆڕینی نووسین بۆ دەنگ بە زمانی کوردی (هەڵبژاردنی زمانی \"ku\" بۆ کوردی) tts = gTTS(text=text, lang='ku', slow=False) tts.save(output_file) os.system(f\"start {output_file}\") # کردنەوەی فایلە دەنگییەکە (لە Windows) # نموونە: text_to_speech_kurdish(\"سڵاو، ئەمە دەنگی منە بە زمانی کوردی.\")","categories":[{"name":"Adversarial","url":"/area/adversarial"},{"name":"Audio","url":"/area/audio"},{"name":"Computer Code","url":"/area/computer-code"},{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Graphs","url":"/area/graphs"},{"name":"Knowledge Base","url":"/area/knowledge-base"},{"name":"Medical","url":"/area/medical"},{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Music","url":"/area/music"},{"name":"Playing Games","url":"/area/playing-games"},{"name":"Robots","url":"/area/robots"},{"name":"Speech","url":"/area/speech"},{"name":"Time Series","url":"/area/time-series"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1419,"papers_with_code":399,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":2},"benchmarks":[{"leaderboard":"/sota/text-to-speech-on","slug":"text-to-speech-on","dataset":".","dataset_url":"/dataset/gun-detection-dataset","rows_in_archive":1,"metrics":["0-shot MRR"],"first_row_in_archive_order":{"model":"mio","paper_title":"StyleTTS 2: Towards Human-Level Text-to-Speech through Style Diffusion and Adversarial Training with Large Speech Language Models","paper_url":"/paper/styletts-2-towards-human-level-text-to-speech","paper_date":"2023-06-13","arxiv_id":"2306.07691","code_links":[{"title":"yl4579/StyleTTS2","url":"https://github.com/yl4579/StyleTTS2"}],"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}}},{"leaderboard":"/sota/text-to-speech-on-1","slug":"text-to-speech-on-1","dataset":"^(#$!@#$)(()))******","dataset_url":null,"rows_in_archive":1,"metrics":["0-shot MRR"],"first_row_in_archive_order":{"model":"Reply","paper_title":"Text + Sketch: Image Compression at Ultra Low Rates","paper_url":"/paper/text-sketch-image-compression-at-ultra-low","paper_date":"2023-07-04","arxiv_id":"2307.01944","code_links":[{"title":"leieric/text-sketch","url":"https://github.com/leieric/text-sketch"}],"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}}}],"datasets":[{"url":"/dataset/gun-detection-dataset","name":"Gun Detection Dataset","full_name":"","num_papers_in_archive":9},{"url":"/dataset/openslr","name":"OpenSLR","full_name":"Open Speech and Language Resources","num_papers_in_archive":2},{"url":"/dataset/arvoice","name":"ArVoice","full_name":"ArVoice: A Multi-Speaker Dataset for Arabic Speech Synthesis","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/10-shot-image-generation","name":"10-shot image generation"},{"url":"/task/audio-visual-speech-recognition","name":"Audio-Visual Speech Recognition"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":399,"tagged_in_all":1419,"items":[{"url":"/paper/wavenet-a-generative-model-for-raw-audio","title":"WaveNet: A Generative Model for Raw Audio","date":"2016-09-12","arxiv_id":"1609.03499","repositories_listed":62,"syntology":{"n":103,"n_ran":41,"n_unverified":62,"n_pointer_only":25}},{"url":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","arxiv_id":"2006.04558","repositories_listed":37,"syntology":{"n":119,"n_ran":79,"n_unverified":40,"n_pointer_only":33}},{"url":"/paper/tacotron-towards-end-to-end-speech-synthesis","title":"Tacotron: Towards End-to-End Speech Synthesis","date":"2017-03-29","arxiv_id":"1703.10135","repositories_listed":30,"syntology":{"n":25,"n_ran":7,"n_unverified":18,"n_pointer_only":6}},{"url":"/paper/fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","arxiv_id":"1905.09263","repositories_listed":22,"syntology":{"n":11,"n_ran":10,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/efficiently-trainable-text-to-speech-system","title":"Efficiently Trainable Text-to-Speech System Based on Deep Convolutional Networks with Guided Attention","date":"2017-10-24","arxiv_id":"1710.08969","repositories_listed":22,"syntology":{"n":28,"n_ran":1,"n_unverified":27,"n_pointer_only":1}},{"url":"/paper/efficient-neural-audio-synthesis","title":"Efficient Neural Audio Synthesis","date":"2018-02-23","arxiv_id":"1802.08435","repositories_listed":16,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/parallel-wavegan-a-fast-waveform-generation","title":"Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram","date":"2019-10-25","arxiv_id":"1910.11480","repositories_listed":12,"syntology":{"n":20,"n_ran":17,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/fastspeech-fastrobustand-controllable-text-to","title":"FastSpeech: Fast,Robustand Controllable Text-to-Speech","date":"2019-05-22","arxiv_id":null,"repositories_listed":11,"syntology":null},{"url":"/paper/transfer-learning-from-speaker-verification","title":"Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis","date":"2018-06-12","arxiv_id":"1806.04558","repositories_listed":11,"syntology":null},{"url":"/paper/diffsinger-diffusion-acoustic-model-for","title":"DiffSinger: Singing Voice Synthesis via Shallow Diffusion Mechanism","date":"2021-05-06","arxiv_id":"2105.02446","repositories_listed":10,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/univnet-a-neural-vocoder-with-multi","title":"UnivNet: A Neural Vocoder with Multi-Resolution Spectrogram Discriminators for High-Fidelity Waveform Generation","date":"2021-06-15","arxiv_id":"2106.07889","repositories_listed":9,"syntology":{"n":12,"n_ran":8,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/robust-universal-neural-vocoding","title":"Robust universal neural vocoding","date":"2018-11-15","arxiv_id":"1811.06292","repositories_listed":8,"syntology":null},{"url":"/paper/neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/deep-voice-3-scaling-text-to-speech-with","title":"Deep Voice 3: Scaling Text-to-Speech with Convolutional Sequence Learning","date":"2017-10-20","arxiv_id":"1710.07654","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/sequence-transduction-with-recurrent-neural","title":"Sequence Transduction with Recurrent Neural Networks","date":"2012-11-14","arxiv_id":"1211.3711","repositories_listed":7,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/soundstream-an-end-to-end-neural-audio-codec","title":"SoundStream: An End-to-End Neural Audio Codec","date":"2021-07-07","arxiv_id":"2107.03312","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","arxiv_id":"2105.06337","repositories_listed":6,"syntology":{"n":18,"n_ran":10,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/non-attentive-tacotron-robust-and-1","title":"Non-Attentive Tacotron: Robust and Controllable Neural TTS Synthesis Including Unsupervised Duration Modeling","date":"2020-10-08","arxiv_id":"2010.04301","repositories_listed":6,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/fastpitch-parallel-text-to-speech-with-pitch","title":"FastPitch: Parallel Text-to-speech with Pitch Prediction","date":"2020-06-11","arxiv_id":"2006.06873","repositories_listed":6,"syntology":{"n":9,"n_ran":8,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/defense-for-black-box-attacks-on-anti","title":"Defense for Black-box Attacks on Anti-spoofing Models by Self-Supervised Learning","date":"2020-06-05","arxiv_id":"2006.03214","repositories_listed":6,"syntology":null},{"url":"/paper/glow-tts-a-generative-flow-for-text-to-speech","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","date":"2020-05-22","arxiv_id":"2005.11129","repositories_listed":6,"syntology":{"n":14,"n_ran":8,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/neural-speech-synthesis-with-transformer","title":"Neural Speech Synthesis with Transformer Network","date":"2018-09-19","arxiv_id":"1809.08895","repositories_listed":6,"syntology":null},{"url":"/paper/melnet-a-generative-model-for-audio-in-the","title":"MelNet: A Generative Model for Audio in the Frequency Domain","date":"2019-06-04","arxiv_id":"1906.01083","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/clarinet-parallel-wave-generation-in-end-to","title":"ClariNet: Parallel Wave Generation in End-to-End Text-to-Speech","date":"2018-07-19","arxiv_id":"1807.07281","repositories_listed":5,"syntology":{"n":13,"n_ran":3,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/statistical-parametric-speech-synthesis-1","title":"Statistical Parametric Speech Synthesis Incorporating Generative Adversarial Networks","date":"2017-09-23","arxiv_id":"1709.08041","repositories_listed":5,"syntology":null},{"url":"/paper/seamlessm4t-massively-multilingual-multimodal","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","date":"2023-08-22","arxiv_id":"2308.11596","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/prodiff-progressive-fast-diffusion-model-for","title":"ProDiff: Progressive Fast Diffusion Model For High-Quality Text-to-Speech","date":"2022-07-13","arxiv_id":"2207.06389","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","repositories_listed":4,"syntology":{"n":12,"n_ran":12,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/fairseq-s-2-a-scalable-and-integrable-speech","title":"fairseq S^2: A Scalable and Integrable Speech Synthesis Toolkit","date":"2021-09-14","arxiv_id":"2109.06912","repositories_listed":4,"syntology":null},{"url":"/paper/lightspeech-lightweight-and-fast-text-to","title":"LightSpeech: Lightweight and Fast Text to Speech with Neural Architecture Search","date":"2021-02-08","arxiv_id":"2102.04040","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":23,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}