{"url":"/dataset/timit","name":"TIMIT","full_name":"TIMIT Acoustic-Phonetic Continuous Speech Corpus","description_markdown":"The **TIMIT** Acoustic-Phonetic Continuous Speech Corpus is a standard dataset used for evaluation of automatic speech recognition systems. It consists of recordings of 630 speakers of 8 dialects of American English each reading 10 phonetically-rich sentences. It also comes with the word and phone-level transcriptions of the speech.\r\n\r\nSource: [Improving neural networks by preventing co-adaptation of feature detectors](https://arxiv.org/abs/1207.0580)\r\nImage Source: [https://roboticrun.wordpress.com/2016/06/21/timit-introduction-the-official-doc/](https://roboticrun.wordpress.com/2016/06/21/timit-introduction-the-official-doc/)","description_withheld":null,"homepage":"https://catalog.ldc.upenn.edu/LDC93S1","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Speech Enhancement","url":"/task/speech-enhancement","datasets_with_task":"/datasets/task/speech-enhancement"},{"name":"Speech Separation","url":"/task/speech-separation","datasets_with_task":"/datasets/task/speech-separation"},{"name":"Speaker-Specific Lip to Speech Synthesis","url":"/task/speaker-specific-lip-to-speech-synthesis","datasets_with_task":"/datasets/task/speaker-specific-lip-to-speech-synthesis"},{"name":"Lip Reading","url":"/task/lip-reading","datasets_with_task":"/datasets/task/lip-reading"},{"name":"Clustering","url":"/task/clustering","datasets_with_task":"/datasets/task/clustering"},{"name":"Phoneme Recognition","url":"/task/phoneme-recognition","datasets_with_task":"/datasets/task/phoneme-recognition"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["timit PER","TIMIT","TCD-TIMIT corpus (mixed-speech)","DARPA TIMIT"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/timit-asr/timit_asr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/timit_asr","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/timit-dataset","frameworks":["tf","pytorch"]}],"num_papers_in_archive":31,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-timit","task":"Speech Recognition","dataset_variant":"TIMIT","rows":22,"metrics":["Percentage error"],"first_row_in_archive_order":{"model":"wav2vec 2.0","paper":"/paper/wav2vec-2-0-a-framework-for-self-supervised","metrics":{"Percentage error":"8.3"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"pytorch/fairseq","url":"https://github.com/pytorch/fairseq"},{"title":"pytorch/fairseq","url":"https://github.com/pytorch/fairseq/tree/master/examples/wav2vec"},{"title":"wenet-e2e/wenet","url":"https://github.com/wenet-e2e/wenet"},{"title":"sh-lee-prml/hierspeechpp","url":"https://github.com/sh-lee-prml/hierspeechpp"},{"title":"facebookresearch/brainmagick","url":"https://github.com/facebookresearch/brainmagick"},{"title":"mailong25/vietnamese-speech-recognition","url":"https://github.com/mailong25/vietnamese-speech-recognition"},{"title":"mailong25/self-supervised-speech-recognition","url":"https://github.com/mailong25/self-supervised-speech-recognition"},{"title":"huseinzol05/malaya-speech","url":"https://github.com/huseinzol05/malaya-speech"},{"title":"neonbjb/ocotillo","url":"https://github.com/neonbjb/ocotillo"},{"title":"shivangi-aneja/FaceTalk","url":"https://github.com/shivangi-aneja/FaceTalk"},{"title":"eastonYi/wav2vec","url":"https://github.com/eastonYi/wav2vec"},{"title":"vasudevgupta7/gsoc-wav2vec2","url":"https://github.com/vasudevgupta7/gsoc-wav2vec2"},{"title":"JoungheeKim/Non-Attentive-Tacotron","url":"https://github.com/JoungheeKim/Non-Attentive-Tacotron"},{"title":"HarunoriKawano/Wav2vec2.0","url":"https://github.com/HarunoriKawano/Wav2vec2.0"},{"title":"gatech-eic/s3-router","url":"https://github.com/gatech-eic/s3-router"},{"title":"BirgerMoell/tmh","url":"https://github.com/BirgerMoell/tmh"},{"title":"liutianlin0121/seislm","url":"https://github.com/liutianlin0121/seislm"},{"title":"AIdeaLab/wav2vec2_docker","url":"https://github.com/AIdeaLab/wav2vec2_docker"},{"title":"nlp-en-es/wav2vec2-spanish","url":"https://github.com/nlp-en-es/wav2vec2-spanish"},{"title":"phanxuanphucnd/wav2asr","url":"https://github.com/phanxuanphucnd/wav2asr"},{"title":"Arizona-Voice/Arizona-spotting","url":"https://github.com/Arizona-Voice/Arizona-spotting"},{"title":"phanxuanphucnd/Arizona-spotting","url":"https://github.com/phanxuanphucnd/Arizona-spotting"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/1/wav2vec2_conformer"},{"title":"phanxuanphucnd/Arizona-asr","url":"https://github.com/phanxuanphucnd/Arizona-asr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/lip-reading-on-tcd-timit-corpus-mixed-speech","task":"Lip Reading","dataset_variant":"TCD-TIMIT corpus (mixed-speech)","rows":1,"metrics":["WER"],"first_row_in_archive_order":{"model":"Lip2Wav","paper":"/paper/learning-individual-speaking-styles-for","metrics":{"WER":"31.26"},"code_links":[{"title":"Rudrabha/Lip2Wav","url":"https://github.com/Rudrabha/Lip2Wav"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speaker-specific-lip-to-speech-synthesis-on-1","task":"Speaker-Specific Lip to Speech Synthesis","dataset_variant":"TCD-TIMIT corpus (mixed-speech)","rows":1,"metrics":["ESTOI","PESQ","STOI"],"first_row_in_archive_order":{"model":"Lip2Wav","paper":"/paper/learning-individual-speaking-styles-for","metrics":{"ESTOI":"36.5","PESQ":"1.35","STOI":"0.558"},"code_links":[{"title":"Rudrabha/Lip2Wav","url":"https://github.com/Rudrabha/Lip2Wav"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-enhancement-on-tcd-timit-corpus-mixed","task":"Speech Enhancement","dataset_variant":"TCD-TIMIT corpus (mixed-speech)","rows":1,"metrics":["PESQ"],"first_row_in_archive_order":{"model":"Audio-Visual concat-ref","paper":"/paper/face-landmark-based-speaker-independent-audio","metrics":{"PESQ":"3.03"},"code_links":[{"title":"dr-pato/audio_visual_speech_enhancement","url":"https://github.com/dr-pato/audio_visual_speech_enhancement"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-separation-on-tcd-timit-corpus-mixed","task":"Speech Separation","dataset_variant":"TCD-TIMIT corpus (mixed-speech)","rows":1,"metrics":["SDR"],"first_row_in_archive_order":{"model":"Audio-Visual concat-ref","paper":"/paper/face-landmark-based-speaker-independent-audio","metrics":{"SDR":"10.55"},"code_links":[{"title":"dr-pato/audio_visual_speech_enhancement","url":"https://github.com/dr-pato/audio_visual_speech_enhancement"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-darpa-timit","task":"Speech Recognition","dataset_variant":"DARPA TIMIT","rows":0,"metrics":["Test CER"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/wav2vec-2-0-a-framework-for-self-supervised","title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations","date":"2020-06-20","rows_on_this_dataset":1,"code_links":25,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":2,"samples_unverified":7,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-individual-speaking-styles-for","title":"Learning Individual Speaking Styles for Accurate Lip to Speech Synthesis","date":"2020-05-17","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":1,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vq-wav2vec-self-supervised-learning-of-1","title":"vq-wav2vec: Self-Supervised Learning of Discrete Speech Representations","date":"2019-10-12","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-model-for-articulatory-features","title":"Attention model for articulatory features detection","date":"2019-07-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/wav2vec-unsupervised-pre-training-for-speech","title":"wav2vec: Unsupervised Pre-training for Speech Recognition","date":"2019-04-11","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-pytorch-kaldi-speech-recognition-toolkit","title":"The PyTorch-Kaldi Speech Recognition Toolkit","date":"2018-11-19","rows_on_this_dataset":8,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/face-landmark-based-speaker-independent-audio","title":"Face Landmark-based Speaker-Independent Audio-Visual Speech Enhancement in Multi-Talker Environments","date":"2018-11-06","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/quaternion-convolutional-neural-networks-for-1","title":"Quaternion Convolutional Neural Networks for End-to-End Automatic Speech Recognition","date":"2018-06-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/long-short-term-memory-and-learning-to-learn","title":"Long short-term memory and learning-to-learn in networks of spiking neurons","date":"2018-03-26","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/light-gated-recurrent-units-for-speech","title":"Light Gated Recurrent Units for Speech Recognition","date":"2018-03-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/online-and-linear-time-attention-by-enforcing","title":"Online and Linear-Time Attention by Enforcing Monotonic Alignments","date":"2017-04-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/segmental-recurrent-neural-networks-for-end","title":"Segmental Recurrent Neural Networks for End-to-end Speech Recognition","date":"2016-03-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/attention-based-models-for-speech-recognition","title":"Attention-Based Models for Speech Recognition","date":"2015-06-24","rows_on_this_dataset":1,"code_links":14,"syntology":null},{"paper":"/paper/speech-recognition-with-deep-recurrent-neural","title":"Speech Recognition with Deep Recurrent Neural Networks","date":"2013-03-22","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":36,"samples_ran":12,"samples_unverified":24,"pointer_only_for_licence":13,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}