{"url":"/dataset/ljspeech","name":"LJSpeech","full_name":"The LJ Speech Dataset","description_markdown":"This is a public domain speech dataset consisting of 13,100 short audio clips of a single speaker reading passages from 7 non-fiction books. A transcription is provided for each clip. Clips vary in length from 1 to 10 seconds and have a total length of approximately 24 hours. The texts were published between 1884 and 1964, and are in the public domain. The audio was recorded in 2016-17 by the LibriVox project and is also in the public domain.\r\n\r\nSource: [The LJ Speech Dataset](https://keithito.com/LJ-Speech-Dataset/)\r\nImage Source: [https://keithito.com/LJ-Speech-Dataset/](https://keithito.com/LJ-Speech-Dataset/)\r\nAudio Source: [https://keithito.com/LJ-Speech-Dataset/](https://keithito.com/LJ-Speech-Dataset/)","description_withheld":null,"homepage":"https://keithito.com/LJ-Speech-Dataset/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":null,"title":"The lj speech dataset","first_author":null,"url":"https://keithito.com/LJ-Speech-Dataset/"},"license":{"name":"Public domain","url":"https://librivox.org/pages/public-domain/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Automatic Speech Recognition","url":"/task/automatic-speech-recognition-2","datasets_with_task":"/datasets/task/automatic-speech-recognition-2"},{"name":"Speech Synthesis","url":"/task/speech-synthesis","datasets_with_task":"/datasets/task/speech-synthesis"},{"name":"Text-To-Speech Synthesis","url":"/task/text-to-speech-synthesis","datasets_with_task":"/datasets/task/text-to-speech-synthesis"},{"name":"Resynthesis","url":"/task/resynthesis","datasets_with_task":"/datasets/task/resynthesis"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["LJSpeech"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/keithito/lj_speech","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/lj_speech","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/ljspeech","frameworks":["tf","jax"]},{"repo":"https://github.com/pytorch/audio","url":"https://pytorch.org/audio/stable/datasets.html#torchaudio.datasets.LJSPEECH","frameworks":["pytorch"]}],"num_papers_in_archive":323,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/text-to-speech-synthesis-on-ljspeech","task":"Text-To-Speech Synthesis","dataset_variant":"LJSpeech","rows":16,"metrics":["Audio Quality MOS","Pleasantness MOS","Word Error Rate (WER)","MOS","WER (%)"],"first_row_in_archive_order":{"model":"NaturalSpeech","paper":"/paper/naturalspeech-end-to-end-text-to-speech","metrics":{"Audio Quality MOS":"4.56"},"code_links":[{"title":"microsoft/NeuralSpeech","url":"https://github.com/microsoft/NeuralSpeech"},{"title":"daniilrobnikov/vits2","url":"https://github.com/daniilrobnikov/vits2"},{"title":"heatz123/naturalspeech","url":"https://github.com/heatz123/naturalspeech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-synthesis-on-ljspeech","task":"Speech Synthesis","dataset_variant":"LJSpeech","rows":4,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"BDDM vocoder","paper":"/paper/bddm-bilateral-denoising-diffusion-models-for-1","metrics":{"Mean Opinion Score":"4.48"},"code_links":[{"title":"tencent-ailab/bddm","url":"https://github.com/tencent-ailab/bddm"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/matcha-tts-a-fast-tts-architecture-with","title":"Matcha-TTS: A fast TTS architecture with conditional flow matching","date":"2023-09-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/overflow-putting-flows-on-top-of-neural","title":"OverFlow: Putting flows on top of neural transducers for better TTS","date":"2022-11-13","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/naturalspeech-end-to-end-text-to-speech","title":"NaturalSpeech: End-to-End Text to Speech Synthesis with Human-Level Quality","date":"2022-05-09","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fastdiff-a-fast-conditional-diffusion-model","title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","date":"2022-04-21","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bddm-bilateral-denoising-diffusion-models-for-1","title":"BDDM: Bilateral Denoising Diffusion Models for Fast and High-Quality Speech Synthesis","date":"2022-03-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-hmms-are-all-you-need-for-high-quality","title":"Neural HMMs are all you need (for high-quality attention-free TTS)","date":"2021-08-30","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/grad-tts-a-diffusion-probabilistic-model-for","title":"Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech","date":"2021-05-13","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":9,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffwave-a-versatile-diffusion-model-for","title":"DiffWave: A Versatile Diffusion Model for Audio Synthesis","date":"2020-09-21","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":33,"samples_ran":20,"samples_unverified":13,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","rows_on_this_dataset":1,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":119,"samples_ran":73,"samples_unverified":46,"pointer_only_for_licence":33,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glow-tts-a-generative-flow-for-text-to-speech","title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","date":"2020-05-22","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":3,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flowtron-an-autoregressive-flow-based","title":"Flowtron: an Autoregressive Flow-based Generative Network for Text-to-Speech Synthesis","date":"2020-05-12","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":4,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fastspeech-fast-robust-and-controllable-text","title":"FastSpeech: Fast, Robust and Controllable Text to Speech","date":"2019-05-22","rows_on_this_dataset":2,"code_links":22,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neural-speech-synthesis-with-transformer","title":"Neural Speech Synthesis with Transformer Network","date":"2018-09-19","rows_on_this_dataset":1,"code_links":6,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":229,"samples_ran":120,"samples_unverified":109,"pointer_only_for_licence":41,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}