{"url":"/dataset/kazakhtts","name":"KazakhTTS","full_name":null,"description_markdown":"**KazakhTTS** is an open-source speech synthesis dataset for Kazakh, a low-resource language spoken by over 13 million people worldwide. The dataset consists of about 91 hours of transcribed audio recordings spoken by two professional speakers (female and male). It is the first publicly available large-scale dataset developed to promote Kazakh text-to-speech (TTS) applications in both academia and industry.","description_withheld":null,"homepage":"https://github.com/IS2AI/Kazakh_TTS","introduced_date":"2021-04-17","introduced_date_note":null,"introduced_by":{"paper":"/paper/kazakhtts-an-open-source-kazakh-text-to","title":"KazakhTTS: An Open-Source Kazakh Text-to-Speech Synthesis Dataset","first_author":"Saida Mussakhojayeva","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Text-To-Speech Synthesis","url":"/task/text-to-speech-synthesis","datasets_with_task":"/datasets/task/text-to-speech-synthesis"}],"languages":[{"name":"Kazakh","url":"/datasets/language/kazakh"}],"variants":["KazakhTTS"],"data_loaders":[{"repo":"https://github.com/IS2AI/Kazakh_TTS","url":"https://github.com/IS2AI/Kazakh_TTS","frameworks":["pytorch"]}],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}