{"url":"/dataset/tuda","name":"TUDA","full_name":null,"description_markdown":"Overall duration per microphone: about 36 hours (31 hrs train / 2.5 hrs dev / 2.5 hrs test)\r\nCount of microphones: 3 (Microsoft Kinect, Yamaha, Samson)\r\nCount of wave-files per microphone: about 14500\r\nOverall count of participations: 180 (130 male / 50 female)","description_withheld":null,"homepage":"https://www.inf.uni-hamburg.de/en/inst/ab/lt/resources/data/acoustic-models.html","introduced_date":"2015-12-11","introduced_date_note":null,"introduced_by":{"paper":"/paper/open-source-german-distant-speech-recognition","title":"Open Source German Distant Speech Recognition: Corpus and Acoustic Model","first_author":"Stephan Radeck-Arneth","url":null},"license":{"name":"CC-BY-4.0","url":"https://creativecommons.org/licenses/by/4.0/"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"}],"languages":[{"name":"German","url":"/datasets/language/german"}],"variants":["TUDA"],"data_loaders":[{"repo":"https://gitlab.com/jaco-assistant/corcua","url":"https://gitlab.com/jaco-assistant/corcua","frameworks":[]}],"num_papers_in_archive":11,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-tuda","task":"Speech Recognition","dataset_variant":"TUDA","rows":9,"metrics":["Test WER"],"first_row_in_archive_order":{"model":"Conformer-Transducer (no LM)","paper":"/paper/automatic-speech-recognition-in-german-a","metrics":{"Test WER":"5.82%"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/improved-open-source-automatic-subtitling-for","title":"Improved Open Source Automatic Subtitling for Lecture Videos","date":"2022-09-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/automatic-speech-recognition-in-german-a","title":"Automatic Speech Recognition in German: A Detailed Error Analysis","date":"2022-08-03","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scribosermo-fast-speech-to-text-models-for","title":"Scribosermo: Fast Speech-to-Text models for German and other Languages","date":"2021-10-15","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/ims-speech-a-speech-to-text-tool","title":"IMS-Speech: A Speech to Text Tool","date":"2019-08-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/open-source-automatic-speech-recognition-for","title":"Open Source Automatic Speech Recognition for German","date":"2018-07-26","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/open-source-german-distant-speech-recognition","title":"Open Source German Distant Speech Recognition: Corpus and Acoustic Model","date":"2015-12-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ctc-segmentation-of-large-corpora-for-german","title":"CTC-Segmentation of Large Corpora for German End-to-end Speech Recognition","date":null,"rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}