{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/a-neural-attention-model-for-speech-command","title":"A neural attention model for speech command recognition","arxiv_id":"1808.08929","date":"2018-08-27","proceeding":null,"authors":["Douglas Coimbra de Andrade","Sabato Leo","Martin Loesener Da Silva Viana","Christoph Bernkopf"],"abstract":"This paper introduces a convolutional recurrent network with attention for\nspeech command recognition. Attention models are powerful tools to improve\nperformance on natural language, image captioning and speech tasks. The\nproposed model establishes a new state-of-the-art accuracy of 94.1% on Google\nSpeech Commands dataset V1 and 94.5% on V2 (for the 20-commands recognition\ntask), while still keeping a small footprint of only 202K trainable parameters.\nResults are compared with previous convolutional implementations on 5 different\ntasks (20 commands recognition (V1 and V2), 12 commands recognition (V1), 35\nword recognition (V1) and left-right (V1)). We show detailed performance\nresults and demonstrate that the proposed attention mechanism not only improves\nperformance but also allows inspecting what regions of the audio were taken\ninto consideration by the network when outputting a given category.","url_abs":"http://arxiv.org/abs/1808.08929v1","url_pdf":"http://arxiv.org/pdf/1808.08929v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/douglas125/SpeechCmdRecognition","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/Arizona-Voice/blossom","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/httttttt/ResNetAtentionBiLSTM","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"gone","observed_at":"2026-09-18","how":"tree_404+repo_404"}},{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/huckiyang/QuantumSpeech-QCNN","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/huckiyang/speech_quantum_dl","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/renyuanL/ry-Speech-commands","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/widzemin/audio_project","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}},{"paper_slug":"a-neural-attention-model-for-speech-command","repo_url":"https://github.com/google-research/google-research/tree/master/kws_streaming","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"tf","reach":null}],"tasks":[{"task_slug":"image-captioning","task_name":"Image Captioning"},{"task_slug":"model","task_name":"model"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/keyword-spotting-on-google-speech-commands","task":"Keyword Spotting","dataset":"Google Speech Commands","model":"Attention RNN","rank_in_archive_order":13,"of":42,"metrics":{"Google Speech Commands V1 12":"95.6","Google Speech Commands V1 2":"99.2","Google Speech Commands V1 20":"94.1","Google Speech Commands V1 35":"94.3","Google Speech Commands V2 12":"96.9","Google Speech Commands V2 2":"99.4","Google Speech Commands V2 20":"94.5","Google Speech Commands V2 35":"93.9"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1808.08929","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.08929"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huckiyang/QuantumSpeech-QCNN","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/renyuanL/ry-Speech-commands","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/douglas125/SpeechCmdRecognition","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/google-research/google-research/tree/master/kws_streaming","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/widzemin/audio_project","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Arizona-Voice/blossom","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/huckiyang/speech_quantum_dl","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/httttttt/ResNetAtentionBiLSTM","reach":{"status":"gone","observed_at":"2026-09-18","how":"tree_404+repo_404"}}],"summary":{"ran_honours":1,"ran_draft_wrong":1,"unverified":1},"by_repo_kind":{"official":{"samples":1,"ran":1,"repositories":1},"listed":{"samples":2,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":2,"samples":[{"code_sha256_prefix":"720014c5cc086010","entry":"normalize","repo":"renyuanL/ry-Speech-commands","repo_kind":"listed","path":"ryRecog03.py","file_url":"https://github.com/renyuanL/ry-Speech-commands/blob/HEAD/ryRecog03.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"GPL-3.0","inline_ok":false,"mcp_get_code":{"code_sha256":"720014c5cc086010"}},{"code_sha256_prefix":"57c9801112b59c24","entry":"predict_audio","repo":"douglas125/SpeechCmdRecognition","repo_kind":"official","path":"recognize_word.py","file_url":"https://github.com/douglas125/SpeechCmdRecognition/blob/HEAD/recognize_word.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"57c9801112b59c24"}},{"code_sha256_prefix":"6fa05e75f626df54","entry":"ryGet1secSpeech","repo":"renyuanL/ry-Speech-commands","repo_kind":"listed","path":"ryRealTimeAsr03.py","file_url":"https://github.com/renyuanL/ry-Speech-commands/blob/HEAD/ryRealTimeAsr03.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"mcp_get_code":{"code_sha256":"6fa05e75f626df54"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}