{"url":"/dataset/europarl-asr","name":"Europarl-ASR","full_name":null,"description_markdown":"Europarl-ASR (EN) is a 1300-hour English-language speech and text corpus of parliamentary debates for (streaming) Automatic Speech Recognition training and benchmarking, speech data filtering and speech data verbatimization, based on European Parliament speeches and their official transcripts (1996-2020). Includes dev-test sets for streaming ASR benchmarking, made up of 18 hours of manually revised speeches. The availability of manual non-verbatim and verbatim transcripts for dev-test speeches makes this corpus also useful for the assessment of automatic filtering and verbatimization techniques. The corpus is released under an open licence at https://www.mllp.upv.es/europarl-asr/\r\n\r\nEuroparl-ASR CONTENTS: [Speech data] 1300 hours of English-language annotated speech data, 3 full sets of timed transcriptions (official non-verbatim, automatically noise-filtered, automatically verbatimized), 18 hours of speech data with both manually revised verbatim transcriptions and official non-verbatim transcriptions, split in 2 independent validation- evaluation partitions for 2 realistic ASR tasks (with vs. without previous knowledge of the speaker); [Text data] 70 million tokens of English-language text data; [Pretrained language models] the Europarl-ASR English-language n-gram language model and vocabulary.","description_withheld":null,"homepage":"https://www.mllp.upv.es/europarl-asr/","introduced_date":"2021-08-30","introduced_date_note":null,"introduced_by":{"paper":"/paper/europarl-asr-a-large-corpus-of-parliamentary","title":"Europarl-ASR: A Large Corpus of Parliamentary Debates for Streaming ASR Benchmarking and Speech Data Filtering/Verbatimization","first_author":"Gonçal V. Garcés Díaz-Munío","url":null},"license":{"name":"Custom","url":"https://www.mllp.upv.es/git-pub/ggarces/Europarl-ASR/src/master/LICENSE"},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Data Augmentation","url":"/task/data-augmentation","datasets_with_task":"/datasets/task/data-augmentation"},{"name":"Benchmarking","url":"/task/benchmarking","datasets_with_task":"/datasets/task/benchmarking"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Europarl-ASR","Europarl-ASR MEP-test","Europarl-ASR EN Guest-test","Europarl-ASR EN MEP-test"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-europarl-asr-en-guest","task":"Speech Recognition","dataset_variant":"Europarl-ASR EN Guest-test","rows":3,"metrics":["WER"],"first_row_in_archive_order":{"model":"United-MedASR (764M)","paper":"/paper/high-precision-medical-speech-recognition","metrics":{"WER":"0.26"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-europarl-asr-en-mep","task":"Speech Recognition","dataset_variant":"Europarl-ASR EN MEP-test","rows":2,"metrics":["WER"],"first_row_in_archive_order":{"model":"mllp_2021_offline_filt","paper":"/paper/europarl-asr-a-large-corpus-of-parliamentary","metrics":{"WER":"7.8"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/high-precision-medical-speech-recognition","title":"High-precision medical speech recognition through synthetic data and semantic correction: UNITED-MEDASR","date":"2024-11-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/europarl-asr-a-large-corpus-of-parliamentary","title":"Europarl-ASR: A Large Corpus of Parliamentary Debates for Streaming ASR Benchmarking and Speech Data Filtering/Verbatimization","date":"2021-08-30","rows_on_this_dataset":4,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}