{"url":"/dataset/libritts","name":"LibriTTS","full_name":"LibriTTS","description_markdown":"**LibriTTS** is a multi-speaker English corpus of approximately 585 hours of read English speech at 24kHz sampling rate, prepared by Heiga Zen with the assistance of Google Speech and Google Brain team members. The LibriTTS corpus is designed for TTS research. It is derived from the original materials (mp3 audio files from LibriVox and text files from Project Gutenberg) of the LibriSpeech corpus. The main differences from the LibriSpeech corpus are listed below:\r\n\r\n- The audio files are at 24kHz sampling rate.\r\n- The speech is split at sentence breaks.\r\n- Both original and normalized texts are included.\r\n- Contextual information (e.g., neighbouring sentences) can be extracted.\r\n- Utterances with significant background noise are excluded.","description_withheld":null,"homepage":"http://www.openslr.org/60","introduced_date":"2019-04-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/libritts-a-corpus-derived-from-librispeech","title":"LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech","first_author":null,"url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Synthesis","url":"/task/speech-synthesis","datasets_with_task":"/datasets/task/speech-synthesis"},{"name":"Text-To-Speech Synthesis","url":"/task/text-to-speech-synthesis","datasets_with_task":"/datasets/task/text-to-speech-synthesis"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["LibriTTS"],"data_loaders":[{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/libritts","frameworks":["tf","jax"]},{"repo":"https://github.com/pytorch/audio","url":"https://pytorch.org/audio/stable/datasets.html#torchaudio.datasets.LIBRITTS","frameworks":["pytorch"]}],"num_papers_in_archive":257,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-synthesis-on-libritts","task":"Speech Synthesis","dataset_variant":"LibriTTS","rows":15,"metrics":["PESQ","M-STFT","MCD","Periodicity","V/UV F1"],"first_row_in_archive_order":{"model":"PeriodWave-Turbo-L","paper":"/paper/accelerating-high-fidelity-waveform","metrics":{"M-STFT":"0.7358","PESQ":"4.454","Periodicity":"0.0528","V/UV F1":"0.9756"},"code_links":[{"title":"sh-lee-prml/periodwave","url":"https://github.com/sh-lee-prml/periodwave"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/accelerating-high-fidelity-waveform","title":"Accelerating High-Fidelity Waveform Generation via Adversarial Flow Matching Optimization","date":"2024-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/periodwave-multi-period-flow-matching-for","title":"PeriodWave: Multi-Period Flow Matching for High-Fidelity Waveform Generation","date":"2024-08-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rfwave-multi-band-rectified-flow-for-audio","title":"RFWave: Multi-band Rectified Flow for Audio Waveform Reconstruction","date":"2024-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/eva-gan-enhanced-various-audio-generation-via","title":"EVA-GAN: Enhanced Various Audio Generation via Scalable Generative Adversarial Networks","date":"2024-01-31","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/bigvsan-enhancing-gan-based-neural-vocoders-1","title":"BigVSAN: Enhancing GAN-based Neural Vocoders with Slicing Adversarial Network","date":"2023-09-06","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/vocos-closing-the-gap-between-time-domain-and","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","date":"2023-06-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":5,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bigvgan-a-universal-neural-vocoder-with-large","title":"BigVGAN: A Universal Neural Vocoder with Large-Scale Training","date":"2022-06-09","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":8,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hifi-gan-generative-adversarial-networks-for","title":"HiFi-GAN: Generative Adversarial Networks for Efficient and High Fidelity Speech Synthesis","date":"2020-10-12","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":17,"samples_unverified":8,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/speaker-conditional-wavernn-towards-universal","title":"Speaker Conditional WaveRNN: Towards Universal Neural Vocoder for Unseen Speaker and Recording Conditions","date":"2020-08-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/waveflow-a-compact-flow-based-model-for-raw-1","title":"WaveFlow: A Compact Flow-based Model for Raw Audio","date":"2019-12-03","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/waveglow-a-flow-based-generative-network-for","title":"WaveGlow: A Flow-based Generative Network for Speech Synthesis","date":"2018-10-31","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":79,"samples_ran":37,"samples_unverified":42,"pointer_only_for_licence":5,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}