{"url":"/task/speech-recognition-1","name":"speech-recognition","slug":"speech-recognition-1","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":5715,"papers_with_code":1277,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/common-voice","name":"Common Voice","full_name":"Common Voice","num_papers_in_archive":449},{"url":"/dataset/voxpopuli","name":"VoxPopuli","full_name":"","num_papers_in_archive":113},{"url":"/dataset/openslr","name":"OpenSLR","full_name":"Open Speech and Language Resources","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":1277,"tagged_in_all":5715,"items":[{"url":"/paper/split-computing-and-early-exiting-for-deep","title":"Split Computing and Early Exiting for Deep Learning Applications: Survey and Research Challenges","date":"2021-03-08","arxiv_id":"2103.04505","repositories_listed":19,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/robust-speech-recognition-via-large-scale-1","title":"Robust Speech Recognition via Large-Scale Weak Supervision","date":"2022-12-06","arxiv_id":"2212.04356","repositories_listed":15,"syntology":{"n":59,"n_ran":5,"n_unverified":54,"n_pointer_only":18}},{"url":"/paper/wavlm-large-scale-self-supervised-pre","title":"WavLM: Large-Scale Self-Supervised Pre-Training for Full Stack Speech Processing","date":"2021-10-26","arxiv_id":"2110.13900","repositories_listed":9,"syntology":null},{"url":"/paper/isynet-convolutional-neural-networks-design","title":"ISyNet: Convolutional Neural Networks design for AI accelerator","date":"2021-09-04","arxiv_id":"2109.01932","repositories_listed":9,"syntology":null},{"url":"/paper/unsupervised-cross-lingual-representation-3","title":"Unsupervised Cross-lingual Representation Learning for Speech Recognition","date":"2020-06-24","arxiv_id":"2006.13979","repositories_listed":8,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","arxiv_id":"2110.07205","repositories_listed":6,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/contextnet-improving-convolutional-neural","title":"ContextNet: Improving Convolutional Neural Networks for Automatic Speech Recognition with Global Context","date":"2020-05-07","arxiv_id":"2005.03191","repositories_listed":6,"syntology":null},{"url":"/paper/deep-gradient-compression-reducing-the","title":"Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training","date":"2017-12-05","arxiv_id":"1712.01887","repositories_listed":6,"syntology":null},{"url":"/paper/a-simple-way-to-initialize-recurrent-networks","title":"A Simple Way to Initialize Recurrent Networks of Rectified Linear Units","date":"2015-04-03","arxiv_id":"1504.00941","repositories_listed":6,"syntology":null},{"url":"/paper/the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/efficient-self-supervised-learning-with","title":"Efficient Self-supervised Learning with Contextualized Target Representations for Vision, Speech and Language","date":"2022-12-14","arxiv_id":"2212.07525","repositories_listed":5,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/unispeech-sat-universal-speech-representation","title":"UniSpeech-SAT: Universal Speech Representation Learning with Speaker Aware Pre-Training","date":"2021-10-12","arxiv_id":"2110.05752","repositories_listed":5,"syntology":null},{"url":"/paper/unispeech-unified-speech-representation","title":"UniSpeech: Unified Speech Representation Learning with Labeled and Unlabeled Data","date":"2021-01-19","arxiv_id":"2101.07597","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/unified-streaming-and-non-streaming-two-pass","title":"Unified Streaming and Non-streaming Two-pass End-to-end Model for Speech Recognition","date":"2020-12-10","arxiv_id":"2012.05481","repositories_listed":5,"syntology":null},{"url":"/paper/stochastic-attention-head-removal-a-simple","title":"Stochastic Attention Head Removal: A simple and effective method for improving Transformer Based ASR Models","date":"2020-11-08","arxiv_id":"2011.04004","repositories_listed":5,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/fairseq-s2t-fast-speech-to-text-modeling-with","title":"fairseq S2T: Fast Speech-to-Text Modeling with fairseq","date":"2020-10-11","arxiv_id":"2010.05171","repositories_listed":5,"syntology":null},{"url":"/paper/transformer-transducer-a-streamable-speech","title":"Transformer Transducer: A Streamable Speech Recognition Model with Transformer Encoders and RNN-T Loss","date":"2020-02-07","arxiv_id":"2002.02562","repositories_listed":5,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/voicefilter-targeted-voice-separation-by","title":"VoiceFilter: Targeted Voice Separation by Speaker-Conditioned Spectrogram Masking","date":"2018-10-11","arxiv_id":"1810.04826","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/an-overview-of-multi-task-learning-in-deep","title":"An Overview of Multi-Task Learning in Deep Neural Networks","date":"2017-06-15","arxiv_id":"1706.05098","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/first-pass-large-vocabulary-continuous-speech","title":"First-Pass Large Vocabulary Continuous Speech Recognition using Bi-Directional Recurrent DNNs","date":"2014-08-12","arxiv_id":"1408.2873","repositories_listed":5,"syntology":null},{"url":"/paper/speech-slytherin-examining-the-performance","title":"Speech Slytherin: Examining the Performance and Efficiency of Mamba for Speech Separation, Recognition, and Synthesis","date":"2024-07-13","arxiv_id":"2407.09732","repositories_listed":4,"syntology":null},{"url":"/paper/scaling-speech-technology-to-1000-languages-1","title":"Scaling Speech Technology to 1,000+ Languages","date":"2023-05-22","arxiv_id":"2305.13516","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/multi-blank-transducers-for-speech","title":"Multi-blank Transducers for Speech Recognition","date":"2022-11-04","arxiv_id":"2211.03541","repositories_listed":4,"syntology":null},{"url":"/paper/self-supervised-learning-with-random","title":"Self-supervised Learning with Random-projection Quantizer for Speech Recognition","date":"2022-02-03","arxiv_id":"2202.01855","repositories_listed":4,"syntology":{"n":15,"n_ran":9,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/fine-tuning-wav2vec2-for-speaker-recognition","title":"Fine-tuning wav2vec2 for speaker recognition","date":"2021-09-30","arxiv_id":"2109.15053","repositories_listed":4,"syntology":null},{"url":"/paper/unsupervised-speech-recognition","title":"Unsupervised Speech Recognition","date":"2021-05-24","arxiv_id":"2105.11084","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/okwugbe-end-to-end-speech-recognition-for-fon","title":"OkwuGbé: End-to-End Speech Recognition for Fon and Igbo","date":"2021-03-13","arxiv_id":"2103.07762","repositories_listed":4,"syntology":null},{"url":"/paper/wenet-production-first-and-production-ready","title":"WeNet: Production oriented Streaming and Non-streaming End-to-End Speech Recognition Toolkit","date":"2021-02-02","arxiv_id":"2102.01547","repositories_listed":4,"syntology":null},{"url":"/paper/multi-modal-dense-video-captioning","title":"Multi-modal Dense Video Captioning","date":"2020-03-17","arxiv_id":"2003.07758","repositories_listed":4,"syntology":null},{"url":"/paper/fully-quantizing-a-simplified-transformer-for","title":"A Simplified Fully Quantized Transformer for End-to-end Speech Recognition","date":"2019-11-09","arxiv_id":"1911.03604","repositories_listed":4,"syntology":null}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}