{"url":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","slug":"automatic-speech-recognition-2","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":3174,"papers_with_code":677,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":22,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/mnist","name":"MNIST","full_name":"","num_papers_in_archive":7651},{"url":"/dataset/librispeech","name":"LibriSpeech","full_name":"","num_papers_in_archive":2361},{"url":"/dataset/common-voice","name":"Common Voice","full_name":"Common Voice","num_papers_in_archive":449},{"url":"/dataset/ljspeech","name":"LJSpeech","full_name":"The LJ Speech Dataset","num_papers_in_archive":323},{"url":"/dataset/fleurs","name":"FLEURS","full_name":"Few-shot Learning Evaluation of Universal Representations of Speech","num_papers_in_archive":141},{"url":"/dataset/voxpopuli","name":"VoxPopuli","full_name":"","num_papers_in_archive":113},{"url":"/dataset/gigaspeech","name":"GigaSpeech","full_name":"","num_papers_in_archive":87},{"url":"/dataset/multilingual-librispeech","name":"Multilingual LibriSpeech","full_name":"MLS","num_papers_in_archive":77},{"url":"/dataset/ted-lium-3","name":"TED-LIUM","full_name":"","num_papers_in_archive":64},{"url":"/dataset/spgispeech","name":"SPGISpeech","full_name":"","num_papers_in_archive":16},{"url":"/dataset/callhome-american-english-speech","name":"CALLHOME American English Speech","full_name":"","num_papers_in_archive":11},{"url":"/dataset/earnings-22","name":"Earnings-22","full_name":"","num_papers_in_archive":10},{"url":"/dataset/italic","name":"ITALIC","full_name":"ITALIC","num_papers_in_archive":4},{"url":"/dataset/multimed","name":"MultiMed","full_name":"","num_papers_in_archive":3},{"url":"/dataset/jam-alt","name":"Jam-ALT","full_name":"JamALT: A Formatting-Aware Lyrics Transcription Benchmark","num_papers_in_archive":2},{"url":"/dataset/msc","name":"MSC","full_name":null,"num_papers_in_archive":2},{"url":"/dataset/npsc","name":"NPSC","full_name":"Norwegian Parliamentary Speech Corpus","num_papers_in_archive":2},{"url":"/dataset/openslr","name":"OpenSLR","full_name":"Open Speech and Language Resources","num_papers_in_archive":2},{"url":"/dataset/berst","name":"BERSt","full_name":"Basic Emotion Random phrase Shouts","num_papers_in_archive":1},{"url":"/dataset/masc","name":"MASC","full_name":"Manually Annotated Sub-Corpus","num_papers_in_archive":1},{"url":"/dataset/medibeng","name":"MediBeng","full_name":"Synthetic Code-Switched Bengali-English Speech Conversations for Healthcare Applications","num_papers_in_archive":1},{"url":"/dataset/storytelling-video-dataset","name":"Video Dataset","full_name":"Storytelling Video Dataset (Russian, Emotion, Gesture, Speech)","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":677,"tagged_in_all":3174,"items":[{"url":"/paper/speech-commands-a-dataset-for-limited","title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","date":"2018-04-09","arxiv_id":"1804.03209","repositories_listed":35,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/specaugment-a-simple-data-augmentation-method","title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition","date":"2019-04-18","arxiv_id":"1904.08779","repositories_listed":30,"syntology":{"n":18,"n_ran":1,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/conformer-convolution-augmented-transformer","title":"Conformer: Convolution-augmented Transformer for Speech Recognition","date":"2020-05-16","arxiv_id":"2005.08100","repositories_listed":25,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":2}},{"url":"/paper/snips-voice-platform-an-embedded-spoken","title":"Snips Voice Platform: an embedded Spoken Language Understanding system for private-by-design voice interfaces","date":"2018-05-25","arxiv_id":"1805.10190","repositories_listed":16,"syntology":{"n":15,"n_ran":1,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/a-baseline-for-detecting-misclassified-and","title":"A Baseline for Detecting Misclassified and Out-of-Distribution Examples in Neural Networks","date":"2016-10-07","arxiv_id":"1610.02136","repositories_listed":14,"syntology":{"n":21,"n_ran":7,"n_unverified":14,"n_pointer_only":6}},{"url":"/paper/speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","arxiv_id":"2110.07205","repositories_listed":6,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/contextnet-improving-convolutional-neural","title":"ContextNet: Improving Convolutional Neural Networks for Automatic Speech Recognition with Global Context","date":"2020-05-07","arxiv_id":"2005.03191","repositories_listed":6,"syntology":null},{"url":"/paper/advances-in-joint-ctc-attention-based-end-to","title":"Advances in Joint CTC-Attention based End-to-End Speech Recognition with a Deep CNN Encoder and RNN-LM","date":"2017-06-08","arxiv_id":"1706.02737","repositories_listed":6,"syntology":null},{"url":"/paper/stochastic-attention-head-removal-a-simple","title":"Stochastic Attention Head Removal: A simple and effective method for improving Transformer Based ASR Models","date":"2020-11-08","arxiv_id":"2011.04004","repositories_listed":5,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/seamlessm4t-massively-multilingual-multimodal","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","date":"2023-08-22","arxiv_id":"2308.11596","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/scaling-speech-technology-to-1000-languages-1","title":"Scaling Speech Technology to 1,000+ Languages","date":"2023-05-22","arxiv_id":"2305.13516","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/multi-blank-transducers-for-speech","title":"Multi-blank Transducers for Speech Recognition","date":"2022-11-04","arxiv_id":"2211.03541","repositories_listed":4,"syntology":null},{"url":"/paper/squeezeformer-an-efficient-transformer-for","title":"Squeezeformer: An Efficient Transformer for Automatic Speech Recognition","date":"2022-06-02","arxiv_id":"2206.00888","repositories_listed":4,"syntology":{"n":49,"n_ran":31,"n_unverified":18,"n_pointer_only":0}},{"url":"/paper/multi-modal-dense-video-captioning","title":"Multi-modal Dense Video Captioning","date":"2020-03-17","arxiv_id":"2003.07758","repositories_listed":4,"syntology":null},{"url":"/paper/fully-quantizing-a-simplified-transformer-for","title":"A Simplified Fully Quantized Transformer for End-to-end Speech Recognition","date":"2019-11-09","arxiv_id":"1911.03604","repositories_listed":4,"syntology":null},{"url":"/paper/bi-directional-lattice-recurrent-neural","title":"Bi-Directional Lattice Recurrent Neural Networks for Confidence Estimation","date":"2018-10-30","arxiv_id":"1810.13024","repositories_listed":4,"syntology":null},{"url":"/paper/audio-adversarial-examples-targeted-attacks","title":"Audio Adversarial Examples: Targeted Attacks on Speech-to-Text","date":"2018-01-05","arxiv_id":"1801.01944","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/state-of-the-art-speech-recognition-with","title":"State-of-the-art Speech Recognition With Sequence-to-Sequence Models","date":"2017-12-05","arxiv_id":"1712.01769","repositories_listed":4,"syntology":null},{"url":"/paper/eesen-end-to-end-speech-recognition-using","title":"EESEN: End-to-End Speech Recognition using Deep RNN Models and WFST-based Decoding","date":"2015-07-29","arxiv_id":"1507.08240","repositories_listed":4,"syntology":null},{"url":"/paper/neural-nilm-deep-neural-networks-applied-to","title":"Neural NILM: Deep Neural Networks Applied to Energy Disaggregation","date":"2015-07-23","arxiv_id":"1507.06594","repositories_listed":4,"syntology":{"n":17,"n_ran":0,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/an-embarrassingly-simple-approach-for-llm","title":"An Embarrassingly Simple Approach for LLM with Strong ASR Capacity","date":"2024-02-13","arxiv_id":"2402.08846","repositories_listed":3,"syntology":null},{"url":"/paper/a-comparative-analysis-between-conformer","title":"A comparative analysis between Conformer-Transducer, Whisper, and wav2vec2 for improving the child speech recognition","date":"2023-11-07","arxiv_id":"2311.04936","repositories_listed":3,"syntology":null},{"url":"/paper/atco2-corpus-a-large-scale-dataset-for","title":"ATCO2 corpus: A Large-Scale Dataset for Research on Automatic Speech Recognition and Natural Language Understanding of Air Traffic Control Communications","date":"2022-11-08","arxiv_id":"2211.04054","repositories_listed":3,"syntology":null},{"url":"/paper/shallow-fusion-of-weighted-finite-state","title":"Shallow Fusion of Weighted Finite-State Transducer and Language Model for Text Normalization","date":"2022-03-29","arxiv_id":"2203.15917","repositories_listed":3,"syntology":null},{"url":"/paper/mt3-multi-task-multitrack-music-transcription-1","title":"MT3: Multi-Task Multitrack Music Transcription","date":"2021-11-04","arxiv_id":"2111.03017","repositories_listed":3,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/fast-rir-fast-neural-diffuse-room-impulse","title":"FAST-RIR: Fast neural diffuse room impulse response generator","date":"2021-10-07","arxiv_id":"2110.04057","repositories_listed":3,"syntology":null},{"url":"/paper/the-sequence-to-sequence-baseline-for-the","title":"The Sequence-to-Sequence Baseline for the Voice Conversion Challenge 2020: Cascading ASR and TTS","date":"2020-10-06","arxiv_id":"2010.02434","repositories_listed":3,"syntology":null},{"url":"/paper/kospeech-open-source-toolkit-for-end-to-end","title":"KoSpeech: Open-Source Toolkit for End-to-End Korean Speech Recognition","date":"2020-09-07","arxiv_id":"2009.03092","repositories_listed":3,"syntology":null},{"url":"/paper/automatic-speech-recognition-benchmark-for","title":"Automatic Speech Recognition Benchmark for Air-Traffic Communications","date":"2020-06-18","arxiv_id":"2006.10304","repositories_listed":3,"syntology":null},{"url":"/paper/distilling-knowledge-from-ensembles-of","title":"Distilling Knowledge from Ensembles of Acoustic Models for Joint CTC-Attention End-to-End Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09310","repositories_listed":3,"syntology":null}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}