{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/speech-robust-bench-a-robustness-benchmark","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","arxiv_id":"2403.07937","date":"2024-03-08","proceeding":null,"authors":["Muhammad A. Shah","David Solans Noguero","Mikko A. Heikkila","Bhiksha Raj","Nicolas Kourtellis"],"abstract":"As Automatic Speech Recognition (ASR) models become ever more pervasive, it is important to ensure that they make reliable predictions under corruptions present in the physical and digital world. We propose Speech Robust Bench (SRB), a comprehensive benchmark for evaluating the robustness of ASR models to diverse corruptions. SRB is composed of 114 input perturbations which simulate an heterogeneous range of corruptions that ASR models may encounter when deployed in the wild. We use SRB to evaluate the robustness of several state-of-the-art ASR models and observe that model size and certain modeling choices such as the use of discrete representations, or self-training appear to be conducive to robustness. We extend this analysis to measure the robustness of ASR models on data from various demographic subgroups, namely English and Spanish speakers, and males and females. Our results revealed noticeable disparities in the model's robustness across subgroups. We believe that SRB will significantly facilitate future research towards robust ASR models, by making it easier to conduct comprehensive and comparable robustness evaluations.","url_abs":"https://arxiv.org/abs/2403.07937v3","url_pdf":"https://arxiv.org/pdf/2403.07937v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"speech-robust-bench-a-robustness-benchmark","repo_url":"https://github.com/ahmedshah1494/speech_robust_bench","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"adversarial-robustness","task_name":"Adversarial Robustness"},{"task_slug":"automatic-speech-recognition-2","task_name":"Automatic Speech Recognition"},{"task_slug":"automatic-speech-recognition","task_name":"Automatic Speech Recognition (ASR)"},{"task_slug":"robust-speech-recognition","task_name":"Robust Speech Recognition"},{"task_slug":"speech-recognition","task_name":"Speech Recognition"},{"task_slug":"speech-recognition-1","task_name":"speech-recognition"}],"methods":[],"datasets_introduced":[{"slug":"speech-robust-bench","name":"Speech Robust Bench","full_name":""}],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2403.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07937"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/aliutkus/speechmetrics","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/ahmedshah1494/speech_robust_bench","reach":null}],"summary":{"ran":5,"ran_honours":1,"ran_draft_wrong":1,"unverified":2},"by_repo_kind":{"official":{"samples":2,"ran":2,"repositories":1},"found_in_text":{"samples":7,"ran":5,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":2,"samples":[{"code_sha256_prefix":"241f784030cde5ff","entry":"calc_cutoffs","repo":"aliutkus/speechmetrics","repo_kind":"found_in_text","path":"speechmetrics/absolute/srmr/srmr.py","file_url":"https://github.com/aliutkus/speechmetrics/blob/HEAD/speechmetrics/absolute/srmr/srmr.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"241f784030cde5ff"}},{"code_sha256_prefix":"b5459ca42363982f","entry":"compute_modulation_cfs","repo":"aliutkus/speechmetrics","repo_kind":"found_in_text","path":"speechmetrics/absolute/srmr/modulation_filters.py","file_url":"https://github.com/aliutkus/speechmetrics/blob/HEAD/speechmetrics/absolute/srmr/modulation_filters.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"b5459ca42363982f"}},{"code_sha256_prefix":"5becc66305cae027","entry":"get_metric_from_adv_file","repo":"ahmedshah1494/speech_robust_bench","repo_kind":"official","path":"collate_results.py","file_url":"https://github.com/ahmedshah1494/speech_robust_bench/blob/HEAD/collate_results.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5becc66305cae027"}},{"code_sha256_prefix":"0042a025d9a06a2a","entry":"hilbert","repo":"aliutkus/speechmetrics","repo_kind":"found_in_text","path":"speechmetrics/absolute/srmr/hilbert.py","file_url":"https://github.com/aliutkus/speechmetrics/blob/HEAD/speechmetrics/absolute/srmr/hilbert.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0042a025d9a06a2a"}},{"code_sha256_prefix":"0e2befbcb003efd5","entry":"load_full_result_from_adv_file","repo":"ahmedshah1494/speech_robust_bench","repo_kind":"official","path":"collate_results.py","file_url":"https://github.com/ahmedshah1494/speech_robust_bench/blob/HEAD/collate_results.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0e2befbcb003efd5"}},{"code_sha256_prefix":"130aff587564956a","entry":"normalize_energy","repo":"aliutkus/speechmetrics","repo_kind":"found_in_text","path":"speechmetrics/absolute/srmr/srmr.py","file_url":"https://github.com/aliutkus/speechmetrics/blob/HEAD/speechmetrics/absolute/srmr/srmr.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"130aff587564956a"}},{"code_sha256_prefix":"77b0d935f3b71ee4","entry":"segment_axis","repo":"aliutkus/speechmetrics","repo_kind":"found_in_text","path":"speechmetrics/absolute/srmr/segmentaxis.py","file_url":"https://github.com/aliutkus/speechmetrics/blob/HEAD/speechmetrics/absolute/srmr/segmentaxis.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"77b0d935f3b71ee4"}},{"code_sha256_prefix":"c047662b19befeeb","entry":"make_modulation_filter","repo":"aliutkus/speechmetrics","repo_kind":"found_in_text","path":"speechmetrics/absolute/srmr/modulation_filters.py","file_url":"https://github.com/aliutkus/speechmetrics/blob/HEAD/speechmetrics/absolute/srmr/modulation_filters.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c047662b19befeeb"}},{"code_sha256_prefix":"8deddaec312d1a99","entry":"modulation_filterbank","repo":"aliutkus/speechmetrics","repo_kind":"found_in_text","path":"speechmetrics/absolute/srmr/modulation_filters.py","file_url":"https://github.com/aliutkus/speechmetrics/blob/HEAD/speechmetrics/absolute/srmr/modulation_filters.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8deddaec312d1a99"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}