{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/attention-based-models-for-speech-recognition","title":"Attention-Based Models for Speech Recognition","arxiv_id":"1506.07503","date":"2015-06-24","proceeding":"NeurIPS 2015 12","authors":["Jan Chorowski","Dzmitry Bahdanau","Dmitriy Serdyuk","Kyunghyun Cho","Yoshua Bengio"],"abstract":"Recurrent sequence generators conditioned on input data through an attention\nmechanism have recently shown very good performance on a range of tasks in-\ncluding machine translation, handwriting synthesis and image caption gen-\neration. We extend the attention-mechanism with features needed for speech\nrecognition. We show that while an adaptation of the model used for machine\ntranslation in reaches a competitive 18.7% phoneme error rate (PER) on the\nTIMIT phoneme recognition task, it can only be applied to utterances which are\nroughly as long as the ones it was trained on. We offer a qualitative\nexplanation of this failure and propose a novel and generic method of adding\nlocation-awareness to the attention mechanism to alleviate this issue. The new\nmethod yields a model that is robust to long inputs and achieves 18% PER in\nsingle utterances and 20% in 10-times longer (repeated) utterances. Finally, we\npropose a change to the at- tention mechanism that prevents it from\nconcentrating too much on single frames, which further reduces PER to 17.6%\nlevel.","url_abs":"http://arxiv.org/abs/1506.07503v1","url_pdf":"http://arxiv.org/pdf/1506.07503v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/30stomercury/Automatic_Speech_Recognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"unanswered"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/Alexander-H-Liu/End-to-end-ASR-Pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/CKRC24/Listen-and-Translate","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/biyoml/End-to-End-Mandarin-ASR","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/biyoml/Pytorch-End-to-End-ASR-on-TIMIT","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/jackjhliu/End-to-End-Mandarin-ASR","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/jackjhliu/Pytorch-End-to-End-ASR-on-TIMIT","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/mnm-rnd/elsa-voice-asr","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/msalhab96/SpeeQ","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/neil-zeng/asr","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/s3prl/End-to-end-ASR-Pytorch","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/sooftware/End-to-end-Speech-Recognition","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/sooftware/OpenSpeech","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"attention-based-models-for-speech-recognition","repo_url":"https://github.com/sooftware/nlp-attentions","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"machine-translation","task_name":"Machine Translation"},{"task_slug":"phoneme-recognition","task_name":"Phoneme Recognition"},{"task_slug":"speech-recognition","task_name":"Speech Recognition"},{"task_slug":"translation","task_name":"Translation"}],"methods":[{"method_slug":"location-sensitive-attention","method_name":"Location Sensitive Attention"},{"method_slug":"tanh-activation","method_name":"Tanh Activation"}],"datasets_introduced":[],"methods_introduced":[{"slug":"location-sensitive-attention","name":"Location Sensitive Attention","full_name":"Location Sensitive Attention"}],"results":[{"leaderboard":"/sota/speech-recognition-on-timit","task":"Speech Recognition","dataset":"TIMIT","model":"Bi-RNN + Attention","rank_in_archive_order":17,"of":22,"metrics":{"Percentage error":"17.6"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1506.07503","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}