{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/deep-speech-scaling-up-end-to-end-speech","title":"Deep Speech: Scaling up end-to-end speech recognition","arxiv_id":"1412.5567","date":"2014-12-17","proceeding":null,"authors":["Awni Hannun","Carl Case","Jared Casper","Bryan Catanzaro","Greg Diamos","Erich Elsen","Ryan Prenger","Sanjeev Satheesh","Shubho Sengupta","Adam Coates","Andrew Y. Ng"],"abstract":"We present a state-of-the-art speech recognition system developed using\nend-to-end deep learning. Our architecture is significantly simpler than\ntraditional speech systems, which rely on laboriously engineered processing\npipelines; these traditional systems also tend to perform poorly when used in\nnoisy environments. In contrast, our system does not need hand-designed\ncomponents to model background noise, reverberation, or speaker variation, but\ninstead directly learns a function that is robust to such effects. We do not\nneed a phoneme dictionary, nor even the concept of a \"phoneme.\" Key to our\napproach is a well-optimized RNN training system that uses multiple GPUs, as\nwell as a set of novel data synthesis techniques that allow us to efficiently\nobtain a large amount of varied data for training. Our system, called Deep\nSpeech, outperforms previously published results on the widely studied\nSwitchboard Hub5'00, achieving 16.0% error on the full test set. Deep Speech\nalso handles challenging noisy environments better than widely used,\nstate-of-the-art commercial speech systems.","url_abs":"http://arxiv.org/abs/1412.5567v2","url_pdf":"http://arxiv.org/pdf/1412.5567v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/PaddlePaddle/PaddleSpeech","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"paddle","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/CorrelAid/codingchallenge1020_team1","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/Digital-Umuganda/Deepspeech-Kinyarwanda","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MPL-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/GeorgeFedoseev/DeepSpeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/IBM/MAX-Speech-to-Text-Converter","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/Loghijiaha/DeepSpeech-Indo","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MPL-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/Picovoice/speech-to-text-benchmark","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/Picovoice/stt-benchmark","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"Apache-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/RashadGarayev/TRSpeech-to-text","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/RezisEwig/unity_speech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/WalterJohnson0/DeepSpeech-KerasRebuild","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/YuBeomGon/DeepSpeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/anssssss/Vietnamese-Speech-Recognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/bjtommychen/Keras_DeepSpeech2_SpeechRecognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/lissyx/STT","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MPL-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/mangushev/deep_speech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/mozilla/DeepSpeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MPL-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/mozilla/STT","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MPL-2.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/msalhab96/SpeeQ","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/myrtleSoftware/deepspeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/pannous/caffe-speech-recognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"caffe2","reach":{"status":"ok"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/robmsmt/KerasDeepSpeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"AGPL-3.0"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/soarsmu/crossasr","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"paddle","reach":{"status":"ok"}},{"paper_slug":"deep-speech-scaling-up-end-to-end-speech","repo_url":"https://github.com/tuanio/deepspeech-ctc","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"accented-speech-recognition","task_name":"Accented Speech Recognition"},{"task_slug":"speech-recognition","task_name":"Speech Recognition"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/accented-speech-recognition-on-voxforge-3","task":"Accented Speech Recognition","dataset":"VoxForge American-Canadian","model":"Deep Speech","rank_in_archive_order":2,"of":2,"metrics":{"Percentage error":"15.01"},"uses_additional_data":false},{"leaderboard":"/sota/accented-speech-recognition-on-voxforge-1","task":"Accented Speech Recognition","dataset":"VoxForge Commonwealth","model":"Deep Speech","rank_in_archive_order":2,"of":2,"metrics":{"Percentage error":"28.46"},"uses_additional_data":false},{"leaderboard":"/sota/accented-speech-recognition-on-voxforge-2","task":"Accented Speech Recognition","dataset":"VoxForge European","model":"Deep Speech","rank_in_archive_order":2,"of":2,"metrics":{"Percentage error":"31.20"},"uses_additional_data":false},{"leaderboard":"/sota/accented-speech-recognition-on-voxforge","task":"Accented Speech Recognition","dataset":"VoxForge Indian","model":"Deep Speech","rank_in_archive_order":2,"of":2,"metrics":{"Percentage error":"45.35"},"uses_additional_data":false},{"leaderboard":"/sota/noisy-speech-recognition-on-chime-clean","task":"Noisy Speech Recognition","dataset":"CHiME clean","model":"CNN + Bi-RNN + CTC (speech to letters)","rank_in_archive_order":2,"of":2,"metrics":{"Percentage error":"6.3"},"uses_additional_data":false},{"leaderboard":"/sota/noisy-speech-recognition-on-chime-real","task":"Noisy Speech Recognition","dataset":"CHiME real","model":"CNN + Bi-RNN + CTC (speech to letters)","rank_in_archive_order":5,"of":5,"metrics":{"Percentage error":"67.94"},"uses_additional_data":false},{"leaderboard":"/sota/speech-recognition-on-switchboard-hub500","task":"Speech Recognition","dataset":"Switchboard + Hub500","model":"Deep Speech + FSH","rank_in_archive_order":20,"of":30,"metrics":{"Percentage error":"12.6"},"uses_additional_data":false},{"leaderboard":"/sota/speech-recognition-on-switchboard-hub500","task":"Speech Recognition","dataset":"Switchboard + Hub500","model":"CNN + Bi-RNN + CTC (speech to letters), 25.9% WER if trainedonlyon SWB","rank_in_archive_order":21,"of":30,"metrics":{"Percentage error":"12.6"},"uses_additional_data":false},{"leaderboard":"/sota/speech-recognition-on-switchboard-hub500","task":"Speech Recognition","dataset":"Switchboard + Hub500","model":"Deep Speech","rank_in_archive_order":30,"of":30,"metrics":{"Percentage error":"20"},"uses_additional_data":false},{"leaderboard":"/sota/speech-recognition-on-swb_hub_500-wer","task":"Speech Recognition","dataset":"swb_hub_500 WER fullSWBCH","model":"CNN + Bi-RNN + CTC (speech to letters), 25.9% WER if trainedonlyon SWB","rank_in_archive_order":8,"of":12,"metrics":{"Percentage error":"16"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=1412.5567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1412.5567"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/WalterJohnson0/DeepSpeech-KerasRebuild","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/robmsmt/KerasDeepSpeech","reach":{"status":"ok","spdx":"AGPL-3.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mozilla/DeepSpeech","reach":{"status":"ok","spdx":"MPL-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/pannous/caffe-speech-recognition","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/anssssss/Vietnamese-Speech-Recognition","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/PaddlePaddle/PaddleSpeech","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Picovoice/speech-to-text-benchmark","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/bjtommychen/Keras_DeepSpeech2_SpeechRecognition","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/GeorgeFedoseev/DeepSpeech","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/RezisEwig/unity_speech","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/YuBeomGon/DeepSpeech","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Picovoice/stt-benchmark","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/soarsmu/crossasr","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/tuanio/deepspeech-ctc","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Digital-Umuganda/Deepspeech-Kinyarwanda","reach":{"status":"ok","spdx":"MPL-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/IBM/MAX-Speech-to-Text-Converter","reach":{"status":"ok","spdx":"Apache-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Loghijiaha/DeepSpeech-Indo","reach":{"status":"ok","spdx":"MPL-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mozilla/STT","reach":{"status":"ok","spdx":"MPL-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/mangushev/deep_speech","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/msalhab96/SpeeQ","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/CorrelAid/codingchallenge1020_team1","reach":{"status":"unanswered"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/lissyx/STT","reach":{"status":"ok","spdx":"MPL-2.0"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/myrtleSoftware/deepspeech","reach":{"status":"ok","spdx":"NOASSERTION"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/RashadGarayev/TRSpeech-to-text","reach":{"status":"ok"}}],"summary":{"ran_honours":2,"ran_fixture":2,"ran_draft_wrong":5},"by_repo_kind":{"listed":{"samples":9,"ran":9,"repositories":3}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":8,"samples":[{"code_sha256_prefix":"51944d6733966676","entry":"get_audio_format","repo":"YuBeomGon/DeepSpeech","repo_kind":"listed","path":"util/audio.py","file_url":"https://github.com/YuBeomGon/DeepSpeech/blob/HEAD/util/audio.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"51944d6733966676"}},{"code_sha256_prefix":"1cb244d6be841554","entry":"get_duration","repo":"YuBeomGon/DeepSpeech","repo_kind":"listed","path":"util/audio.py","file_url":"https://github.com/YuBeomGon/DeepSpeech/blob/HEAD/util/audio.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"1cb244d6be841554"}},{"code_sha256_prefix":"50176dcc6ddee70e","entry":"get_num_samples","repo":"YuBeomGon/DeepSpeech","repo_kind":"listed","path":"util/audio.py","file_url":"https://github.com/YuBeomGon/DeepSpeech/blob/HEAD/util/audio.py","link_basis":"first_harvest_node","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"50176dcc6ddee70e"}},{"code_sha256_prefix":"7c01db5a56271dd6","entry":"out","repo":"anssssss/Vietnamese-Speech-Recognition","repo_kind":"listed","path":"src/evaluate.py","file_url":"https://github.com/anssssss/Vietnamese-Speech-Recognition/blob/HEAD/src/evaluate.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"7c01db5a56271dd6"}},{"code_sha256_prefix":"232015401802af14","entry":"process","repo":"anssssss/Vietnamese-Speech-Recognition","repo_kind":"listed","path":"src/model.py","file_url":"https://github.com/anssssss/Vietnamese-Speech-Recognition/blob/HEAD/src/model.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"232015401802af14"}},{"code_sha256_prefix":"a7e204ec063f026b","entry":"read_labels","repo":"mangushev/deep_speech","repo_kind":"listed","path":"prepare_libri.py","file_url":"https://github.com/mangushev/deep_speech/blob/HEAD/prepare_libri.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"a7e204ec063f026b"}},{"code_sha256_prefix":"3306aa443c436614","entry":"resolve","repo":"YuBeomGon/DeepSpeech","repo_kind":"listed","path":"transcribe.py","file_url":"https://github.com/YuBeomGon/DeepSpeech/blob/HEAD/transcribe.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"3306aa443c436614"}},{"code_sha256_prefix":"fb9b291fc581e945","entry":"sparse_tensor_value_to_texts","repo":"YuBeomGon/DeepSpeech","repo_kind":"listed","path":"evaluate.py","file_url":"https://github.com/YuBeomGon/DeepSpeech/blob/HEAD/evaluate.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"fb9b291fc581e945"}},{"code_sha256_prefix":"970f4ac6b4021aaf","entry":"sparse_tuple_to_texts","repo":"YuBeomGon/DeepSpeech","repo_kind":"listed","path":"evaluate.py","file_url":"https://github.com/YuBeomGon/DeepSpeech/blob/HEAD/evaluate.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"970f4ac6b4021aaf"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}