{"url":"/task/robust-speech-recognition","name":"Robust Speech Recognition","slug":"robust-speech-recognition","description_markdown":null,"categories":[{"name":"Adversarial","url":"/area/adversarial"},{"name":"Audio","url":"/area/audio"},{"name":"Computer Code","url":"/area/computer-code"},{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Medical","url":"/area/medical"},{"name":"Methodology","url":"/area/methodology"},{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Music","url":"/area/music"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Reasoning","url":"/area/reasoning"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":97,"papers_with_code":26,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/dipco","name":"DiPCo","full_name":"DiPCo -- Dinner Party Corpus","num_papers_in_archive":16},{"url":"/dataset/google-speech-commands-musan","name":"Google Speech Commands - Musan","full_name":"","num_papers_in_archive":3},{"url":"/dataset/fsc-p2","name":"FSC-P2","full_name":"Fearless Steps Challenge Phase2","num_papers_in_archive":1},{"url":"/dataset/speech-robust-bench","name":"Speech Robust Bench","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/speech-recognition","name":"Speech Recognition"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":26,"of":26,"tagged_in_all":97,"items":[{"url":"/paper/robust-speech-recognition-via-large-scale-1","title":"Robust Speech Recognition via Large-Scale Weak Supervision","date":"2022-12-06","arxiv_id":"2212.04356","repositories_listed":15,"syntology":{"n":59,"n_ran":5,"n_unverified":54,"n_pointer_only":18}},{"url":"/paper/interactive-feature-fusion-for-end-to-end","title":"Interactive Feature Fusion for End-to-End Noise-Robust Speech Recognition","date":"2021-10-11","arxiv_id":"2110.05267","repositories_listed":2,"syntology":null},{"url":"/paper/scalable-factorized-hierarchical-variational","title":"Scalable Factorized Hierarchical Variational Autoencoder Training","date":"2018-04-09","arxiv_id":"1804.03201","repositories_listed":2,"syntology":null},{"url":"/paper/very-deep-convolutional-neural-networks-for-1","title":"Very Deep Convolutional Neural Networks for Robust Speech Recognition","date":"2016-10-02","arxiv_id":"1610.00277","repositories_listed":2,"syntology":null},{"url":"/paper/dysarthria-normalization-via-local-lie-group","title":"Dysarthria Normalization via Local Lie Group Transformations for Robust ASR","date":"2025-04-16","arxiv_id":"2504.12279","repositories_listed":1,"syntology":null},{"url":"/paper/mms-llama-efficient-llm-based-audio-visual-1","title":"MMS-LLaMA: Efficient LLM-based Audio-Visual Speech Recognition with Minimal Multimodal Speech Tokens","date":"2025-03-14","arxiv_id":"2503.11315","repositories_listed":1,"syntology":null},{"url":"/paper/mwhisper-flamingo-for-multilingual-audio","title":"mWhisper-Flamingo for Multilingual Audio-Visual Noise-Robust Speech Recognition","date":"2025-02-03","arxiv_id":"2502.01547","repositories_listed":1,"syntology":null},{"url":"/paper/channel-aware-domain-adaptive-generative","title":"Channel-Aware Domain-Adaptive Generative Adversarial Network for Robust Speech Recognition","date":"2024-09-19","arxiv_id":"2409.12386","repositories_listed":1,"syntology":null},{"url":"/paper/speech-robust-bench-a-robustness-benchmark","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","date":"2024-03-08","arxiv_id":"2403.07937","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/large-language-models-are-efficient-learners","title":"Large Language Models are Efficient Learners of Noise-Robust Speech Recognition","date":"2024-01-19","arxiv_id":"2401.10446","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/lyricwhiz-robust-multilingual-zero-shot","title":"LyricWhiz: Robust Multilingual Zero-shot Lyrics Transcription by Whispering to ChatGPT","date":"2023-06-29","arxiv_id":"2306.17103","repositories_listed":1,"syntology":null},{"url":"/paper/muavic-a-multilingual-audio-visual-corpus-for","title":"MuAViC: A Multilingual Audio-Visual Corpus for Robust Speech Recognition and Robust Speech-to-Text Translation","date":"2023-03-01","arxiv_id":"2303.00628","repositories_listed":1,"syntology":null},{"url":"/paper/gradient-remedy-for-multi-task-learning-in","title":"Gradient Remedy for Multi-Task Learning in End-to-End Noise-Robust Speech Recognition","date":"2023-02-22","arxiv_id":"2302.11362","repositories_listed":1,"syntology":null},{"url":"/paper/audio-visual-efficient-conformer-for-robust","title":"Audio-Visual Efficient Conformer for Robust Speech Recognition","date":"2023-01-04","arxiv_id":"2301.01456","repositories_listed":1,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/ccc-wav2vec-2-0-clustering-aided-cross","title":"CCC-wav2vec 2.0: Clustering aided Cross Contrastive Self-supervised learning of speech representations","date":"2022-10-05","arxiv_id":"2210.02592","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/dent-ddsp-data-efficient-noisy-speech","title":"DENT-DDSP: Data-efficient noisy speech generator using differentiable digital signal processors for explicit distortion modelling and noise-robust speech recognition","date":"2022-08-01","arxiv_id":"2208.00987","repositories_listed":1,"syntology":null},{"url":"/paper/espnet-se-speech-enhancement-for-robust","title":"ESPnet-SE++: Speech Enhancement for Robust Speech Recognition, Translation, and Understanding","date":"2022-07-19","arxiv_id":"2207.09514","repositories_listed":1,"syntology":null},{"url":"/paper/dual-path-style-learning-for-end-to-end-noise","title":"Dual-Path Style Learning for End-to-End Noise-Robust Speech Recognition","date":"2022-03-28","arxiv_id":"2203.14838","repositories_listed":1,"syntology":null},{"url":"/paper/speech-enhanced-and-noise-aware-networks-for","title":"Speech-enhanced and Noise-aware Networks for Robust Speech Recognition","date":"2022-03-25","arxiv_id":"2203.13696","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-randomized-smoothing-for-1","title":"Sequential Randomized Smoothing for Adversarially Robust Speech Recognition","date":"2021-11-05","arxiv_id":"2112.03000","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-end-to-end-models-for","title":"An Investigation of End-to-End Models for Robust Speech Recognition","date":"2021-02-11","arxiv_id":"2102.06237","repositories_listed":1,"syntology":null},{"url":"/paper/domain-adaptation-using-class-similarity-for","title":"Domain Adaptation Using Class Similarity for Robust Speech Recognition","date":"2020-11-05","arxiv_id":"2011.02782","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-self-supervised-learning-for-1","title":"Multi-task self-supervised learning for Robust Speech Recognition","date":"2020-01-25","arxiv_id":"2001.09239","repositories_listed":1,"syntology":null},{"url":"/paper/parzen-filters-for-spectral-decomposition-of","title":"Learning Waveform-Based Acoustic Models using Deep Variational Convolutional Neural Networks","date":"2019-06-23","arxiv_id":"1906.09526","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-speech-domain-adaptation-based","title":"Unsupervised Speech Domain Adaptation Based on Disentangled Representation Learning for Robust Speech Recognition","date":"2019-04-12","arxiv_id":"1904.06086","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-generative-adversarial-networks","title":"Investigating Generative Adversarial Networks based Speech Dereverberation for Robust Speech Recognition","date":"2018-03-27","arxiv_id":"1803.10132","repositories_listed":1,"syntology":null}],"syntology_records":5,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}