{"url":"/dataset/the-spoken-wikipedia-corpora","name":"The Spoken Wikipedia Corpora","full_name":"The Spoken Wikipedia Corpora","description_markdown":"The SWC is a corpus of aligned Spoken Wikipedia articles from the English, German, and Dutch Wikipedia. This corpus has several outstanding characteristics:\r\n\r\n- hundreds of hours of aligned audio\r\n- from a diverse set of readers\r\n- about a diverse set of topics\r\n- in a well-researched textual genre\r\n- licensed under a free license (CC BY-SA 4.0)\r\n- Annotations can be mapped back to the original html\r\n- phoneme-level alignments","description_withheld":null,"homepage":"https://nats.gitlab.io/swc/","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":{"name":"CC BY-SA 4.0","url":"https://creativecommons.org/licenses/by-sa/4.0/"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Automatic Speech Recognition (ASR)","url":"/task/automatic-speech-recognition","datasets_with_task":"/datasets/task/automatic-speech-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"German","url":"/datasets/language/german"},{"name":"Dutch","url":"/datasets/language/dutch"}],"variants":["The Spoken Wikipedia Corpora"],"data_loaders":[{"repo":"https://gitlab.com/jaco-assistant/corcua","url":"https://gitlab.com/jaco-assistant/corcua","frameworks":[]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/automatic-speech-recognition-on-the-spoken","task":"Automatic Speech Recognition (ASR)","dataset_variant":"The Spoken Wikipedia Corpora","rows":1,"metrics":["WER (%)"],"first_row_in_archive_order":{"model":"Conformer Transducer","paper":"/paper/automatic-speech-recognition-in-german-a","metrics":{"WER (%)":"8.04%"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/automatic-speech-recognition-in-german-a","title":"Automatic Speech Recognition in German: A Detailed Error Analysis","date":"2022-08-03","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}