{"url":"/sota/speech-recognition-on-aishell-1","task":{"name":"Speech Recognition","url":"/task/speech-recognition","note":null},"dataset":{"name":"AISHELL-1","url":"/dataset/aishell-1"},"category":"Audio","categories":["Audio","Speech"],"category_note":null,"description":"**Speech Recognition** is the task of converting spoken language into text. It involves recognizing the words spoken in an audio recording and transcribing them into a written format. The goal is to accurately transcribe the speech in real-time or from recorded audio, taking into account factors such as accents, speaking speed, and background noise.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [SpecAugment](https://arxiv.org/pdf/1904.08779v2.pdf) )</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Word Error Rate (WER)","Params(M)"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Word Error Rate (WER)":"lower","Params(M)":"lower"}},"counts":{"rows":18,"rows_with_code":15,"rows_with_paper_page":18,"rows_dated":18,"rows_using_additional_data":4},"rows":[{"rank_in_archive_order":1,"model":"FireRedASR-AED","metrics":{"Params(M)":"1,100","Word Error Rate (WER)":"0.55"},"uses_additional_data":true,"paper_date":"2025-01-24","paper":"/paper/fireredasr-open-source-industrial-grade","paper_url":"https://arxiv.org/abs/2501.14350v1","paper_title":"FireRedASR: Open-Source Industrial-Grade Mandarin Speech Recognition Models from Encoder-Decoder to LLM Integration","code":"https://github.com/fireredteam/fireredasr","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"Seed-ASR","metrics":{"Word Error Rate (WER)":"0.68"},"uses_additional_data":true,"paper_date":"2024-07-05","paper":"/paper/seed-asr-understanding-diverse-speech-and","paper_url":"https://arxiv.org/abs/2407.04675v2","paper_title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":3,"model":"Qwen-Audio","metrics":{"Word Error Rate (WER)":"1.29"},"uses_additional_data":true,"paper_date":"2023-11-14","paper":"/paper/qwen-audio-advancing-universal-audio","paper_url":"https://arxiv.org/abs/2311.07919v2","paper_title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","code":"https://github.com/alibaba-damo-academy/FunASR","n_code_links":2,"syntology":{"n_ran":5,"n_unverified":2,"n_samples":7,"n_pointer_only_licence":7}},{"rank_in_archive_order":4,"model":"MMSpeech With LM","metrics":{"Word Error Rate (WER)":"1.9"},"uses_additional_data":false,"paper_date":"2022-11-29","paper":"/paper/mmspeech-multi-modal-multi-task-encoder","paper_url":"https://arxiv.org/abs/2212.00500v1","paper_title":"MMSpeech: Multi-modal Multi-task Encoder-Decoder Pre-training for Speech Recognition","code":"https://github.com/ofa-sys/ofa","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"Paraformer-large","metrics":{"Params(M)":"220","Word Error Rate (WER)":"1.95"},"uses_additional_data":true,"paper_date":"2023-05-18","paper":"/paper/funasr-a-fundamental-end-to-end-speech","paper_url":"https://arxiv.org/abs/2305.11013v1","paper_title":"FunASR: A Fundamental End-to-End Speech Recognition Toolkit","code":"https://github.com/alibaba-damo-academy/FunASR","n_code_links":1,"syntology":null},{"rank_in_archive_order":6,"model":"Zipformer+CR-CTC (no external language model)","metrics":{"Params(M)":"66.2","Word Error Rate (WER)":"4.02"},"uses_additional_data":false,"paper_date":"2024-10-07","paper":"/paper/cr-ctc-consistency-regularization-on-ctc-for","paper_url":"https://arxiv.org/abs/2410.05101v4","paper_title":"CR-CTC: Consistency regularization on CTC for improved speech recognition","code":"https://github.com/k2-fsa/icefall","n_code_links":1,"syntology":null},{"rank_in_archive_order":7,"model":"Lightweight Transducer With LM","metrics":{"Params(M)":"45.3","Word Error Rate (WER)":"4.03"},"uses_additional_data":false,"paper_date":"2024-09-05","paper":"/paper/lightweight-transducer-based-on-frame-level","paper_url":"https://arxiv.org/abs/2409.13698v2","paper_title":"Lightweight Transducer Based on Frame-Level Criterion","code":"https://github.com/wangmengzhi/Lightweight-Transducer","n_code_links":1,"syntology":null},{"rank_in_archive_order":8,"model":"SE-WSBO With LM","metrics":{"Params(M)":"46","Word Error Rate (WER)":"4.1"},"uses_additional_data":false,"paper_date":"2022-07-24","paper":"/paper/improving-mandarin-speech-recogntion-with","paper_url":"https://arxiv.org/abs/2207.11697v5","paper_title":"Improving Mandarin Speech Recogntion with Block-augmented Transformer","code":"https://github.com/LeonWlw/asr_blockformer","n_code_links":2,"syntology":null},{"rank_in_archive_order":9,"model":"CIF-HKD With LM","metrics":{"Params(M)":"47","Word Error Rate (WER)":"4.1"},"uses_additional_data":false,"paper_date":"2023-01-30","paper":"/paper/knowledge-transfer-from-pre-trained-language","paper_url":"https://arxiv.org/abs/2301.13003v2","paper_title":"Knowledge Transfer from Pre-trained Language Models to Cif-based Speech Recognizers via Hierarchical Distillation","code":"https://github.com/MingLunHan/CIF-PyTorch","n_code_links":2,"syntology":null},{"rank_in_archive_order":10,"model":"Lightweight Transducer","metrics":{"Params(M)":"45.3","Word Error Rate (WER)":"4.31"},"uses_additional_data":false,"paper_date":"2024-09-05","paper":"/paper/lightweight-transducer-based-on-frame-level","paper_url":"https://arxiv.org/abs/2409.13698v2","paper_title":"Lightweight Transducer Based on Frame-Level Criterion","code":"https://github.com/wangmengzhi/Lightweight-Transducer","n_code_links":1,"syntology":null},{"rank_in_archive_order":11,"model":"UMA","metrics":{"Params(M)":"44.7","Word Error Rate (WER)":"4.7"},"uses_additional_data":false,"paper_date":"2023-09-15","paper":"/paper/unimodal-aggregation-for-ctc-based-speech","paper_url":"https://arxiv.org/abs/2309.08150v2","paper_title":"Unimodal Aggregation for CTC-based Speech Recognition","code":"https://github.com/Audio-WestlakeU/UMA-ASR","n_code_links":1,"syntology":null},{"rank_in_archive_order":12,"model":"U2","metrics":{"Params(M)":"47","Word Error Rate (WER)":"4.72"},"uses_additional_data":false,"paper_date":"2020-12-10","paper":"/paper/unified-streaming-and-non-streaming-two-pass","paper_url":"https://arxiv.org/abs/2012.05481v2","paper_title":"Unified Streaming and Non-streaming Two-pass End-to-end Model for Speech Recognition","code":"https://github.com/PaddlePaddle/PaddleSpeech","n_code_links":5,"syntology":null},{"rank_in_archive_order":13,"model":"Paraformer","metrics":{"Params(M)":"46.3","Word Error Rate (WER)":"4.95"},"uses_additional_data":false,"paper_date":"2023-05-18","paper":"/paper/funasr-a-fundamental-end-to-end-speech","paper_url":"https://arxiv.org/abs/2305.11013v1","paper_title":"FunASR: A Fundamental End-to-End Speech Recognition Toolkit","code":"https://github.com/alibaba-damo-academy/FunASR","n_code_links":1,"syntology":null},{"rank_in_archive_order":14,"model":"BAT","metrics":{"Params(M)":"90","Word Error Rate (WER)":"4.97"},"uses_additional_data":false,"paper_date":"2023-05-19","paper":"/paper/bat-boundary-aware-transducer-for-memory","paper_url":"https://arxiv.org/abs/2305.11571v1","paper_title":"BAT: Boundary aware transducer for memory-efficient and low-latency ASR","code":"https://github.com/alibaba-damo-academy/FunASR","n_code_links":1,"syntology":null},{"rank_in_archive_order":15,"model":"CTC-CRF 4gram-LM","metrics":{"Word Error Rate (WER)":"6.34"},"uses_additional_data":false,"paper_date":"2020-05-27","paper":"/paper/cat-a-ctc-crf-based-asr-toolkit-bridging-the","paper_url":"https://arxiv.org/abs/2005.13326v2","paper_title":"CAT: A CTC-CRF based ASR Toolkit Bridging the Hybrid and the End-to-end Approaches towards Data Efficiency and Low Latency","code":"https://github.com/thu-spmi/cat","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"BRA-E","metrics":{"Params(M)":"8.5","Word Error Rate (WER)":"6.63"},"uses_additional_data":false,"paper_date":"2023-03-23","paper":"/paper/beyond-universal-transformer-block-reusing","paper_url":"https://arxiv.org/abs/2303.13072v2","paper_title":"Beyond Universal Transformer: block reusing with adaptor in Transformer for automatic speech recognition","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":17,"model":"CTC/Att","metrics":{"Word Error Rate (WER)":"6.7"},"uses_additional_data":false,"paper_date":"2019-09-13","paper":"/paper/a-comparative-study-on-transformer-vs-rnn-in","paper_url":"https://arxiv.org/abs/1909.06317v2","paper_title":"A Comparative Study on Transformer vs RNN in Speech Applications","code":"https://github.com/espnet/espnet","n_code_links":2,"syntology":null},{"rank_in_archive_order":18,"model":"Att","metrics":{"Word Error Rate (WER)":"18.7"},"uses_additional_data":false,"paper_date":"2018-08-30","paper":"/paper/end-to-end-speech-recognition-with-adaptive","paper_url":"http://arxiv.org/abs/1808.10088v2","paper_title":"End-to-end Speech Recognition with Adaptive Computation Steps","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":1,"rows_with_any_sample_ran":1,"distinct_papers_with_graph_line":1,"distinct_papers_with_any_sample_ran":1,"samples_over_distinct_papers":{"n_ran":5,"n_unverified":2,"n_samples":7,"n_pointer_only_licence":7,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":5,"n_unverified":2,"n_samples":7,"n_pointer_only_licence":7,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}