{"url":"/dataset/mediaspeech","name":"MediaSpeech","full_name":null,"description_markdown":"**MediaSpeech** is a media speech dataset (you might have guessed this) built with the purpose of testing Automated Speech Recognition (ASR) systems performance. The dataset consists of short speech segments automatically extracted from media videos available on YouTube and manually transcribed, with some pre- and post-processing. The dataset contains 10 hours of speech for each language provided. This release contains audio datasets in French, Arabic, Turkish and Spanish, and is a part of a larger private dataset.\r\n\r\nSource: [MediaSpeech: Multilanguage ASR Benchmark and Dataset](https://github.com/NTRLab/MediaSpeech)","description_withheld":null,"homepage":"https://github.com/NTRLab/MediaSpeech","introduced_date":"2021-03-30","introduced_date_note":null,"introduced_by":{"paper":"/paper/mediaspeech-multilanguage-asr-benchmark-and","title":"MediaSpeech: Multilanguage ASR Benchmark and Dataset","first_author":"Rostislav Kolobov","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Spanish","url":"/datasets/language/spanish"}],"variants":["MediaSpeech"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-mediaspeech","task":"Speech Recognition","dataset_variant":"MediaSpeech","rows":8,"metrics":["WER for Arabic","WER for French","WER for Spanish","WER for Turkish"],"first_row_in_archive_order":{"model":"Quartznet","paper":"/paper/mediaspeech-multilanguage-asr-benchmark-and","metrics":{"WER for Arabic":"0.1300","WER for French":"0.1915","WER for Spanish":"0.1826","WER for Turkish":"0.1422"},"code_links":[{"title":"NTRLab/MediaSpeech","url":"https://github.com/NTRLab/MediaSpeech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mediaspeech-multilanguage-asr-benchmark-and","title":"MediaSpeech: Multilanguage ASR Benchmark and Dataset","date":"2021-03-30","rows_on_this_dataset":8,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}