{"url":"/dataset/indictts","name":"IndicTTS","full_name":null,"description_markdown":"A special corpus of Indian languages covering 13 major languages of India. It comprises of 10000+ spoken sentences/utterances each of mono and English recorded by both Male and Female native speakers. Speech waveform files are available in .wav format along with the corresponding text. We hope that these recordings will be useful for researchers and speech technologists working on synthesis and recognition. You can request zip archives of the entire database here.","description_withheld":null,"homepage":"https://www.iitm.ac.in/donlab/tts/index.php","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Spoken language identification","url":"/task/spoken-language-identification","datasets_with_task":"/datasets/task/spoken-language-identification"},{"name":"Speech Synthesis - Gujarati","url":"/task/speech-synthesis-gujarati","datasets_with_task":"/datasets/task/speech-synthesis-gujarati"},{"name":"Speech Synthesis - Tamil","url":"/task/speech-synthesis-tamil","datasets_with_task":"/datasets/task/speech-synthesis-tamil"},{"name":"Speech Synthesis - Kannada","url":"/task/speech-synthesis-kannada","datasets_with_task":"/datasets/task/speech-synthesis-kannada"},{"name":"Speech Synthesis - Malayalam","url":"/task/speech-synthesis-malayalam","datasets_with_task":"/datasets/task/speech-synthesis-malayalam"},{"name":"Speech Synthesis - Telugu","url":"/task/speech-synthesis-telugu","datasets_with_task":"/datasets/task/speech-synthesis-telugu"},{"name":"Speech Synthesis - Assamese","url":"/task/speech-synthesis-assamese","datasets_with_task":"/datasets/task/speech-synthesis-assamese"},{"name":"Speech Synthesis - Bengali","url":"/task/speech-synthesis-bengali","datasets_with_task":"/datasets/task/speech-synthesis-bengali"},{"name":"Speech Synthesis - Bodo","url":"/task/speech-synthesis-bodo","datasets_with_task":"/datasets/task/speech-synthesis-bodo"},{"name":"Speech Synthesis - Hindi","url":"/task/speech-synthesis-hindi","datasets_with_task":"/datasets/task/speech-synthesis-hindi"},{"name":"Speech Synthesis - Manipuri","url":"/task/speech-synthesis-manipuri","datasets_with_task":"/datasets/task/speech-synthesis-manipuri"},{"name":"Speech Synthesis - Marathi","url":"/task/speech-synthesis-marathi","datasets_with_task":"/datasets/task/speech-synthesis-marathi"},{"name":"Speech Synthesis - Rajasthani","url":"/task/speech-synthesis-rajasthani","datasets_with_task":"/datasets/task/speech-synthesis-rajasthani"}],"languages":[{"name":"Bengali","url":"/datasets/language/bengali"},{"name":"Hindi","url":"/datasets/language/hindi"},{"name":"Marathi","url":"/datasets/language/marathi"},{"name":"Tamil","url":"/datasets/language/tamil"},{"name":"Telugu","url":"/datasets/language/telugu"},{"name":"Assamese","url":"/datasets/language/assamese"},{"name":"Bodo (India)","url":"/datasets/language/bodo-india"},{"name":"Gujarati","url":"/datasets/language/gujarati"},{"name":"Kannada","url":"/datasets/language/kannada"},{"name":"Malayalam","url":"/datasets/language/malayalam"},{"name":"Manipuri","url":"/datasets/language/manipuri"},{"name":"Rajasthani","url":"/datasets/language/rajasthani"}],"variants":["IndicTTS"],"data_loaders":[],"num_papers_in_archive":3,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/spoken-language-identification-on-indictts","task":"Spoken language identification","dataset_variant":"IndicTTS","rows":3,"metrics":["Classification Accuracy"],"first_row_in_archive_order":{"model":"CRNN","paper":"/paper/is-attention-always-needed-a-case-study-on","metrics":{"Classification Accuracy":"0.987"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-assamese-on-indictts","task":"Speech Synthesis - Assamese","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"2.39"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-bengali-on-indictts","task":"Speech Synthesis - Bengali","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.37"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-bodo-on-indictts","task":"Speech Synthesis - Bodo","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.53"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-gujarati-on-indictts","task":"Speech Synthesis - Gujarati","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.58"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-hindi-on-indictts","task":"Speech Synthesis - Hindi","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"4.00"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-kannada-on-indictts","task":"Speech Synthesis - Kannada","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.68"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-malayalam-on-indictts","task":"Speech Synthesis - Malayalam","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.64"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-manipuri-on-indictts","task":"Speech Synthesis - Manipuri","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.30"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-marathi-on-indictts","task":"Speech Synthesis - Marathi","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.26"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-rajasthani-on-indictts","task":"Speech Synthesis - Rajasthani","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.40"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-tamil-on-indictts","task":"Speech Synthesis - Tamil","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.84"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-telugu-on-indictts","task":"Speech Synthesis - Telugu","dataset_variant":"IndicTTS","rows":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"AI4BharatTTS - FastPitch with HiFiGAN","paper":"/paper/towards-building-text-to-speech-systems-for","metrics":{"Mean Opinion Score":"3.66"},"code_links":[{"title":"ai4bharat/indic-tts","url":"https://github.com/ai4bharat/indic-tts"},{"title":"gokulkarthik/text2speech","url":"https://github.com/gokulkarthik/text2speech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/towards-building-text-to-speech-systems-for","title":"Towards Building Text-To-Speech Systems for the Next Billion Users","date":"2022-11-17","rows_on_this_dataset":12,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/is-attention-always-needed-a-case-study-on","title":"Is Attention always needed? A Case Study on Language Identification from Speech","date":"2021-10-05","rows_on_this_dataset":3,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}