{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/an-end-to-end-architecture-for-keyword","title":"An End-to-End Architecture for Keyword Spotting and Voice Activity Detection","arxiv_id":"1611.09405","date":"2016-11-28","proceeding":null,"authors":["Chris Lengerich","Awni Hannun"],"abstract":"We propose a single neural network architecture for two tasks: on-line\nkeyword spotting and voice activity detection. We develop novel inference\nalgorithms for an end-to-end Recurrent Neural Network trained with the\nConnectionist Temporal Classification loss function which allow our model to\nachieve high accuracy on both keyword spotting and voice activity detection\nwithout retraining. In contrast to prior voice activity detection models, our\narchitecture does not require aligned training data and uses the same\nparameters as the keyword spotting model. This allows us to deploy a high\nquality voice activity detector with no additional memory or maintenance\nrequirements.","url_abs":"http://arxiv.org/abs/1611.09405v1","url_pdf":"http://arxiv.org/pdf/1611.09405v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"an-end-to-end-architecture-for-keyword","repo_url":"https://github.com/magahub/mind","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"an-end-to-end-architecture-for-keyword","repo_url":"https://github.com/mindorii/kws","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null},{"paper_slug":"an-end-to-end-architecture-for-keyword","repo_url":"https://github.com/taylorlu/AudioKWS","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null}],"tasks":[{"task_slug":"action-detection","task_name":"Action Detection"},{"task_slug":"activity-detection","task_name":"Activity Detection"},{"task_slug":"classification","task_name":"General Classification"},{"task_slug":"keyword-spotting","task_name":"Keyword Spotting"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":null,"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}