{"url":"/sota/speech-recognition-on-timit","task":{"name":"Speech Recognition","url":"/task/speech-recognition","note":null},"dataset":{"name":"TIMIT","url":"/dataset/timit"},"category":"Audio","categories":["Audio","Speech"],"category_note":null,"description":"**Speech Recognition** is the task of converting spoken language into text. It involves recognizing the words spoken in an audio recording and transcribing them into a written format. The goal is to accurately transcribe the speech in real-time or from recorded audio, taking into account factors such as accents, speaking speed, and background noise.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [SpecAugment](https://arxiv.org/pdf/1904.08779v2.pdf) )</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Percentage error"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Percentage error":"lower"}},"counts":{"rows":22,"rows_with_code":19,"rows_with_paper_page":20,"rows_dated":20,"rows_using_additional_data":3},"rows":[{"rank_in_archive_order":1,"model":"wav2vec 2.0","metrics":{"Percentage error":"8.3"},"uses_additional_data":true,"paper_date":"2020-06-20","paper":"/paper/wav2vec-2-0-a-framework-for-self-supervised","paper_url":"https://arxiv.org/abs/2006.11477v3","paper_title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations","code":"https://github.com/huggingface/transformers","n_code_links":25,"syntology":{"n_ran":2,"n_unverified":7,"n_samples":9,"n_pointer_only_licence":2}},{"rank_in_archive_order":2,"model":"vq-wav2vec","metrics":{"Percentage error":"11.6"},"uses_additional_data":true,"paper_date":"2019-10-12","paper":"/paper/vq-wav2vec-self-supervised-learning-of-1","paper_url":"https://arxiv.org/abs/1910.05453v3","paper_title":"vq-wav2vec: Self-Supervised Learning of Discrete Speech Representations","code":"https://github.com/pytorch/fairseq","n_code_links":3,"syntology":{"n_ran":0,"n_unverified":3,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":3,"model":"LiGRU + Dropout + BatchNorm + Monophone Reg","metrics":{"Percentage error":"14.2"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":4,"model":"LSTM + Dropout + BatchNorm + Monophone Reg","metrics":{"Percentage error":"14.5"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":5,"model":"wav2vec","metrics":{"Percentage error":"14.7"},"uses_additional_data":true,"paper_date":"2019-04-11","paper":"/paper/wav2vec-unsupervised-pre-training-for-speech","paper_url":"https://arxiv.org/abs/1904.05862v4","paper_title":"wav2vec: Unsupervised Pre-training for Speech Recognition","code":"https://github.com/pytorch/fairseq","n_code_links":7,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"GRU + Dropout + BatchNorm + Monophone Reg","metrics":{"Percentage error":"14.9"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":7,"model":"Li-GRU + fMLLR features","metrics":{"Percentage error":"14.9"},"uses_additional_data":false,"paper_date":"2018-03-26","paper":"/paper/light-gated-recurrent-units-for-speech","paper_url":"http://arxiv.org/abs/1803.10225v1","paper_title":"Light Gated Recurrent Units for Speech Recognition","code":"https://github.com/mravanelli/theano-kaldi-rnn","n_code_links":1,"syntology":null},{"rank_in_archive_order":8,"model":"RNN + Dropout + BatchNorm + Monophone Reg","metrics":{"Percentage error":"15.9"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":9,"model":"LSTM","metrics":{"Percentage error":"16.0"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":10,"model":"Li-GRU","metrics":{"Percentage error":"16.3"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":11,"model":"Hierarchical maxout CNN + Dropout","metrics":{"Percentage error":"16.5"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":12,"model":"RNN","metrics":{"Percentage error":"16.5"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":13,"model":"GRU","metrics":{"Percentage error":"16.6"},"uses_additional_data":false,"paper_date":"2018-11-19","paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","paper_url":"http://arxiv.org/abs/1811.07453v2","paper_title":"The PyTorch-Kaldi Speech Recognition Toolkit","code":"https://github.com/mravanelli/pytorch-kaldi","n_code_links":11,"syntology":{"n_ran":5,"n_unverified":1,"n_samples":6,"n_pointer_only_licence":6}},{"rank_in_archive_order":14,"model":"CNN in time and frequency + dropout, 17.6% w/o dropout","metrics":{"Percentage error":"16.7"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":15,"model":"Light Gated Recurrent Units","metrics":{"Percentage error":"16.7"},"uses_additional_data":false,"paper_date":"2018-03-26","paper":"/paper/light-gated-recurrent-units-for-speech","paper_url":"http://arxiv.org/abs/1803.10225v1","paper_title":"Light Gated Recurrent Units for Speech Recognition","code":"https://github.com/mravanelli/theano-kaldi-rnn","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"RNN-CRF on 24(x3) MFSC","metrics":{"Percentage error":"17.3"},"uses_additional_data":false,"paper_date":"2016-03-01","paper":"/paper/segmental-recurrent-neural-networks-for-end","paper_url":"http://arxiv.org/abs/1603.00223v2","paper_title":"Segmental Recurrent Neural Networks for End-to-end Speech Recognition","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":17,"model":"Bi-RNN + Attention","metrics":{"Percentage error":"17.6"},"uses_additional_data":false,"paper_date":"2015-06-24","paper":"/paper/attention-based-models-for-speech-recognition","paper_url":"http://arxiv.org/abs/1506.07503v1","paper_title":"Attention-Based Models for Speech Recognition","code":"https://github.com/Alexander-H-Liu/End-to-end-ASR-Pytorch","n_code_links":14,"syntology":null},{"rank_in_archive_order":18,"model":"Bi-LSTM + skip connections w/ CTC","metrics":{"Percentage error":"17.7"},"uses_additional_data":false,"paper_date":"2013-03-22","paper":"/paper/speech-recognition-with-deep-recurrent-neural","paper_url":"http://arxiv.org/abs/1303.5778v1","paper_title":"Speech Recognition with Deep Recurrent Neural Networks","code":"https://github.com/HawkAaron/warp-transducer","n_code_links":5,"syntology":{"n_ran":0,"n_unverified":3,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":19,"model":"QCNN-10L-256FM","metrics":{"Percentage error":"19.64"},"uses_additional_data":false,"paper_date":"2018-06-20","paper":"/paper/quaternion-convolutional-neural-networks-for-1","paper_url":"http://arxiv.org/abs/1806.07789v1","paper_title":"Quaternion Convolutional Neural Networks for End-to-End Automatic Speech Recognition","code":"https://github.com/Riccardo-Vecchi/Pytorch-Quaternion-Neural-Networks","n_code_links":1,"syntology":null},{"rank_in_archive_order":20,"model":"Soft Monotonic Attention (ours, offline)","metrics":{"Percentage error":"20.1"},"uses_additional_data":false,"paper_date":"2017-04-03","paper":"/paper/online-and-linear-time-attention-by-enforcing","paper_url":"http://arxiv.org/abs/1704.00784v2","paper_title":"Online and Linear-Time Attention by Enforcing Monotonic Alignments","code":"https://github.com/craffel/mad","n_code_links":2,"syntology":null},{"rank_in_archive_order":21,"model":"LAS multitask with indicators sampling","metrics":{"Percentage error":"20.4"},"uses_additional_data":false,"paper_date":"2019-07-02","paper":"/paper/attention-model-for-articulatory-features","paper_url":"https://arxiv.org/abs/1907.01914v1","paper_title":"Attention model for articulatory features detection","code":"https://github.com/sciforce/phones-las","n_code_links":1,"syntology":null},{"rank_in_archive_order":22,"model":"LSNN","metrics":{"Percentage error":"33.2"},"uses_additional_data":false,"paper_date":"2018-03-26","paper":"/paper/long-short-term-memory-and-learning-to-learn","paper_url":"http://arxiv.org/abs/1803.09574v4","paper_title":"Long short-term memory and learning-to-learn in networks of spiking neurons","code":"https://github.com/norse/norse","n_code_links":2,"syntology":{"n_ran":3,"n_unverified":2,"n_samples":5,"n_pointer_only_licence":5}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":13,"rows_with_any_sample_ran":11,"distinct_papers_with_graph_line":6,"distinct_papers_with_any_sample_ran":4,"samples_over_distinct_papers":{"n_ran":11,"n_unverified":16,"n_samples":27,"n_pointer_only_licence":13,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":46,"n_unverified":23,"n_samples":69,"n_pointer_only_licence":55,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}