{"url":"/dataset/aishell-1","name":"AISHELL-1","full_name":null,"description_markdown":"AISHELL-1 is a corpus for speech recognition research and building speech recognition systems for Mandarin. \r\n\r\nSource: [AISHELL-1: An Open-Source Mandarin Speech Corpus and A Speech Recognition Baseline](/paper/aishell-1-an-open-source-mandarin-speech)","description_withheld":null,"homepage":"http://www.openslr.org/33/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/aishell-1-an-open-source-mandarin-speech","title":"AISHELL-1: An Open-Source Mandarin Speech Corpus and A Speech Recognition Baseline","first_author":"Hui Bu","url":null},"license":{"name":"Apache-2","url":"http://www.openslr.org/33/"},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"}],"languages":[{"name":"Mandarin Chinese","url":"/datasets/language/mandarin-chinese"}],"variants":["AISHELL-1"],"data_loaders":[],"num_papers_in_archive":197,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-aishell-1","task":"Speech Recognition","dataset_variant":"AISHELL-1","rows":18,"metrics":["Word Error Rate (WER)","Params(M)"],"first_row_in_archive_order":{"model":"FireRedASR-AED","paper":"/paper/fireredasr-open-source-industrial-grade","metrics":{"Params(M)":"1,100","Word Error Rate (WER)":"0.55"},"code_links":[{"title":"fireredteam/fireredasr","url":"https://github.com/fireredteam/fireredasr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/fireredasr-open-source-industrial-grade","title":"FireRedASR: Open-Source Industrial-Grade Mandarin Speech Recognition Models from Encoder-Decoder to LLM Integration","date":"2025-01-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cr-ctc-consistency-regularization-on-ctc-for","title":"CR-CTC: Consistency regularization on CTC for improved speech recognition","date":"2024-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lightweight-transducer-based-on-frame-level","title":"Lightweight Transducer Based on Frame-Level Criterion","date":"2024-09-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/seed-asr-understanding-diverse-speech-and","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","date":"2024-07-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/qwen-audio-advancing-universal-audio","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","date":"2023-11-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unimodal-aggregation-for-ctc-based-speech","title":"Unimodal Aggregation for CTC-based Speech Recognition","date":"2023-09-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bat-boundary-aware-transducer-for-memory","title":"BAT: Boundary aware transducer for memory-efficient and low-latency ASR","date":"2023-05-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/funasr-a-fundamental-end-to-end-speech","title":"FunASR: A Fundamental End-to-End Speech Recognition Toolkit","date":"2023-05-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/beyond-universal-transformer-block-reusing","title":"Beyond Universal Transformer: block reusing with adaptor in Transformer for automatic speech recognition","date":"2023-03-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/knowledge-transfer-from-pre-trained-language","title":"Knowledge Transfer from Pre-trained Language Models to Cif-based Speech Recognizers via Hierarchical Distillation","date":"2023-01-30","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/mmspeech-multi-modal-multi-task-encoder","title":"MMSpeech: Multi-modal Multi-task Encoder-Decoder Pre-training for Speech Recognition","date":"2022-11-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-mandarin-speech-recogntion-with","title":"Improving Mandarin Speech Recogntion with Block-augmented Transformer","date":"2022-07-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/unified-streaming-and-non-streaming-two-pass","title":"Unified Streaming and Non-streaming Two-pass End-to-end Model for Speech Recognition","date":"2020-12-10","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/cat-a-ctc-crf-based-asr-toolkit-bridging-the","title":"CAT: A CTC-CRF based ASR Toolkit Bridging the Hybrid and the End-to-end Approaches towards Data Efficiency and Low Latency","date":"2020-05-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-comparative-study-on-transformer-vs-rnn-in","title":"A Comparative Study on Transformer vs RNN in Speech Applications","date":"2019-09-13","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/end-to-end-speech-recognition-with-adaptive","title":"End-to-end Speech Recognition with Adaptive Computation Steps","date":"2018-08-30","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}