{"url":"/dataset/multilingual-librispeech","name":"Multilingual LibriSpeech","full_name":"MLS","description_markdown":"Multilingual LibriSpeech is a large multilingual corpus suitable for speech research. The dataset is derived from read audiobooks from LibriVox and consists of 8 languages - English, German, Dutch, Spanish, French, Italian, Portuguese, Polish. It includes about 44.5K hours of English and a total of about 6K hours for other languages. \r\n\r\nSource: [MLS: A Large-Scale Multilingual Dataset for Speech Research](/paper/mls-a-large-scale-multilingual-dataset-for)","description_withheld":null,"homepage":"https://github.com/facebookresearch/wav2letter/tree/master/recipes/mls","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/mls-a-large-scale-multilingual-dataset-for","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","first_author":"Vineel Pratap","url":null},"license":{"name":"CC BY 4.0","url":null},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Automatic Speech Recognition","url":"/task/automatic-speech-recognition-2","datasets_with_task":"/datasets/task/automatic-speech-recognition-2"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["facebook/multilingual_librispeech","Multilingual LibriSpeech, short-form (<= 30sec)","multilingual_librispeech es","multilingual_librispeech","facebook/multilingual_librispeech german","facebook/multilingual_librispeech italian","facebook/multilingual_librispeech French","facebook/multilingual_librispeech spanish","Multilingual LibriSpeech (MLS)","Multilingual LibriSpeech"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ntt123/mls-eng-128kb","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/facebook/multilingual_librispeech","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/parler-tts/mls_eng","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/parler-tts/mls-eng-10k-tags_tagged_10k_generated","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/parler-tts/mls_eng_10k","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/pharaouk/mls-eng-10k-tags_tagged_10k_generated","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/parler-tts/mls-eng-speaker-descriptions","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/Tavish9/LiveScene","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/wav2letter","url":"https://github.com/facebookresearch/wav2letter","frameworks":[]}],"num_papers_in_archive":77,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-facebook-multilingual","task":"Speech Recognition","dataset_variant":"facebook/multilingual_librispeech german","rows":1,"metrics":["WER"],"first_row_in_archive_order":{"model":"TDT 0-4","paper":"/paper/efficient-sequence-transduction-by-jointly","metrics":{"WER":"3.93"},"code_links":[{"title":"NVIDIA/NeMo","url":"https://github.com/NVIDIA/NeMo"},{"title":"chimechallenge/C8DASR-Baseline-NeMo","url":"https://github.com/chimechallenge/C8DASR-Baseline-NeMo"},{"title":"kehanlu/Nemo","url":"https://github.com/kehanlu/Nemo"},{"title":"wd929/NeMo","url":"https://github.com/wd929/NeMo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-multilingual-1","task":"Speech Recognition","dataset_variant":"Multilingual LibriSpeech","rows":0,"metrics":["Dev WER","Test WER"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/efficient-sequence-transduction-by-jointly","title":"Efficient Sequence Transduction by Jointly Predicting Tokens and Durations","date":"2023-04-13","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}