{"url":"/task/keyword-spotting","name":"Keyword Spotting","slug":"keyword-spotting","description_markdown":"In speech processing, keyword spotting deals with the identification of keywords in utterances.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Simon Grest](https://github.com/simongrest/kaggle-freesound-audio-tagging-2019) )</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":407,"papers_with_code":113,"benchmarks":10,"benchmark_tables_in_archive":10,"benchmark_tables_shown":10,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":2,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/keyword-spotting-on-quesst","slug":"keyword-spotting-on-quesst","dataset":"QUESST","dataset_url":null,"rows_in_archive":69,"metrics":["Cnxe","MinCnxe","ATWV","MTWV","SSF","PMUs","ISF","PMUi","PL","lowerbound "],"first_row_in_archive_order":{"model":"ELiRF Fusion+Length(All Queries)","paper_title":"ELiRF at MediaEval 2014: Query by Example Search on Speech Task (QUESST)","paper_url":"/paper/elirf-at-mediaeval-2014-query-by-example","paper_date":"2014-10-16","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-google-speech-commands","slug":"keyword-spotting-on-google-speech-commands","dataset":"Google Speech Commands","dataset_url":"/dataset/speech-commands","rows_in_archive":42,"metrics":["Google Speech Commands V1 12","Google Speech Commands V2 12","Google Speech Commands V2 2","Google Speech Commands V2 20","Google Speech Commands V2 35","Google Speech Commands V1 2","Google Speech Commands V1 20","Google Speech Commands V1 35","Google Speech Commands V1 6","10-keyword Speech Commands dataset","Google Speech Command-Musan","% Test Accuracy","Google Speech Commands"],"first_row_in_archive_order":{"model":"TripletLoss-res15","paper_title":"Learning Efficient Representations for Keyword Spotting with Triplet Loss","paper_url":"/paper/learning-efficient-representations-for-3","paper_date":"2021-01-12","arxiv_id":"2101.04792","code_links":[{"title":"roman-vygon/triplet_loss_kws","url":"https://github.com/roman-vygon/triplet_loss_kws"}],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-hey-siri","slug":"keyword-spotting-on-hey-siri","dataset":"hey Siri","dataset_url":null,"rows_in_archive":4,"metrics":["Error Rate"],"first_row_in_archive_order":{"model":"HEiMDaL","paper_title":"HEiMDaL: Highly Efficient Method for Detection and Localization of wake-words","paper_url":"/paper/heimdal-highly-efficient-method-for-detection","paper_date":"2022-10-26","arxiv_id":"2210.15425","code_links":[],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-fkd","slug":"keyword-spotting-on-fkd","dataset":"FKD","dataset_url":"/dataset/fkd","rows_in_archive":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Res26","paper_title":"EfficientNet-Absolute Zero for Continuous Speech Keyword Spotting","paper_url":"/paper/efficientnet-absolute-zero-for-continuous","paper_date":"2020-12-31","arxiv_id":"2012.15695","code_links":[{"title":"AmirmohammadRostami/KeywordsSpotting-EfficientNet-A0","url":"https://github.com/AmirmohammadRostami/KeywordsSpotting-EfficientNet-A0"},{"title":"AmirmohammadRostami/ASV-anti-spoofing-with-EABN","url":"https://github.com/AmirmohammadRostami/ASV-anti-spoofing-with-EABN"}],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-google-speech-commands-v2-3","slug":"keyword-spotting-on-google-speech-commands-v2-3","dataset":"Google Speech Commands V2 35","dataset_url":null,"rows_in_archive":2,"metrics":["Accuracy (10-fold)"],"first_row_in_archive_order":{"model":"QuaternionNeuralNetwork","paper_title":"Towards on-Device Keyword Spotting using Low-Footprint Quaternion Neural Models","paper_url":"/paper/towards-on-device-keyword-spotting-using-low","paper_date":"2023-09-15","arxiv_id":null,"code_links":[{"title":"DataSenseiAryan/GoogleSpeechCommandLowFootprint","url":"https://github.com/DataSenseiAryan/GoogleSpeechCommandLowFootprint"}],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-tau-urban-acoustic-scenes","slug":"keyword-spotting-on-tau-urban-acoustic-scenes","dataset":"TAU Urban Acoustic Scenes 2019","dataset_url":"/dataset/tau-urban-acoustic-scenes-2019","rows_in_archive":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CP-ResNet(ch64) w/ SSN(S=2, A=Sub)","paper_title":"SubSpectral Normalization for Neural Audio Data Processing","paper_url":"/paper/subspectral-normalization-for-neural-audio","paper_date":"2021-03-25","arxiv_id":"2103.13620","code_links":[],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-tensorflow","slug":"keyword-spotting-on-tensorflow","dataset":"TensorFlow","dataset_url":null,"rows_in_archive":2,"metrics":["TFMA"],"first_row_in_archive_order":{"model":"TensorFlow's model version 2","paper_title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","paper_url":"/paper/speech-commands-a-dataset-for-limited","paper_date":"2018-04-09","arxiv_id":"1804.03209","code_links":[{"title":"retrocirce/hts-audio-transformer","url":"https://github.com/retrocirce/hts-audio-transformer"},{"title":"qute012/Wav2Keyword","url":"https://github.com/qute012/Wav2Keyword"},{"title":"Snapchat/snapml-templates","url":"https://github.com/Snapchat/snapml-templates"},{"title":"saraalemadi/DroneAudioDataset","url":"https://github.com/saraalemadi/DroneAudioDataset"},{"title":"Qualcomm-AI-research/bcresnet","url":"https://github.com/Qualcomm-AI-research/bcresnet"},{"title":"TomVeniat/SANAS","url":"https://github.com/TomVeniat/SANAS"},{"title":"ncsoft/phonmatchnet","url":"https://github.com/ncsoft/phonmatchnet"},{"title":"jarfo/gcommands","url":"https://github.com/jarfo/gcommands"},{"title":"idiap/sparch","url":"https://github.com/idiap/sparch"},{"title":"tae-jun/sampleaudio","url":"https://github.com/tae-jun/sampleaudio"},{"title":"phanxuanphucnd/wav2kws","url":"https://github.com/phanxuanphucnd/wav2kws"},{"title":"diggerdu/audiorwkv","url":"https://github.com/diggerdu/audiorwkv"},{"title":"zhenghuatan/audio-adversarial-examples","url":"https://github.com/zhenghuatan/audio-adversarial-examples"},{"title":"diggerdu/AudioMamba","url":"https://github.com/diggerdu/AudioMamba"},{"title":"maxtimer97/ssm-inspired-lif","url":"https://github.com/maxtimer97/ssm-inspired-lif"},{"title":"dapeter/nas-for-kws","url":"https://github.com/dapeter/nas-for-kws"},{"title":"rf5/simple-speech-commands","url":"https://github.com/rf5/simple-speech-commands"},{"title":"Alpe6825/Unicornn","url":"https://github.com/Alpe6825/Unicornn"},{"title":"pspratling/Speech-Recognition","url":"https://github.com/pspratling/Speech-Recognition"},{"title":"pulp-platform/eae-kws","url":"https://github.com/pulp-platform/eae-kws"},{"title":"bozliu/E2E-Keyword-Spotting","url":"https://github.com/bozliu/E2E-Keyword-Spotting"},{"title":"sankalp2610/Speech_Command_Recognition","url":"https://github.com/sankalp2610/Speech_Command_Recognition"},{"title":"toopster/dsc-capstone-project","url":"https://github.com/toopster/dsc-capstone-project"},{"title":"Cbhihe/classify_sound_mobile","url":"https://github.com/Cbhihe/classify_sound_mobile"},{"title":"lukesin/nas-for-kws-2","url":"https://github.com/lukesin/nas-for-kws-2"},{"title":"akommini/Spoken-Numeric-Digit-detection","url":"https://github.com/akommini/Spoken-Numeric-Digit-detection"},{"title":"ayyucedemirbas/gcommands_12_classes","url":"https://github.com/ayyucedemirbas/gcommands_12_classes"},{"title":"widzemin/audio_project","url":"https://github.com/widzemin/audio_project"},{"title":"ROBOTICSENGINEER/End_to_End_Learning_of_Speech_2D_Feature_Trajectory","url":"https://github.com/ROBOTICSENGINEER/End_to_End_Learning_of_Speech_2D_Feature_Trajectory"},{"title":"TMarquet/speech_recognition","url":"https://github.com/TMarquet/speech_recognition"},{"title":"FraCorti/Deep_Subnetworks_for_Dynamic_Resource_Constraints","url":"https://github.com/FraCorti/Deep_Subnetworks_for_Dynamic_Resource_Constraints"},{"title":"Abdallah-Hesham99/tell_your_motor_speed","url":"https://github.com/Abdallah-Hesham99/tell_your_motor_speed"},{"title":"fracorti/reds","url":"https://github.com/fracorti/reds"},{"title":"Benja1972/Speech_recognition_Tensorflow2.0","url":"https://github.com/Benja1972/Speech_recognition_Tensorflow2.0"},{"title":"unibe-cns/gle-code","url":"https://github.com/unibe-cns/gle-code"}],"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":1}}},{"leaderboard":"/sota/keyword-spotting-on-voxforge","slug":"keyword-spotting-on-voxforge","dataset":"VoxForge","dataset_url":"/dataset/voxforge","rows_in_archive":2,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"1D-ConvNet","paper_title":"Spoken Language Identification using ConvNets","paper_url":"/paper/spoken-language-identification-using-convnets","paper_date":"2019-10-09","arxiv_id":"1910.04269","code_links":[],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-google-speech-commands-v2-1","slug":"keyword-spotting-on-google-speech-commands-v2-1","dataset":"Google Speech Commands V2 12","dataset_url":null,"rows_in_archive":1,"metrics":["Accuracy","Latency (STM32F746ZG)"],"first_row_in_archive_order":{"model":"MicroNet-KWS-L","paper_title":"MicroNets: Neural Network Architectures for Deploying TinyML Applications on Commodity Microcontrollers","paper_url":"/paper/micronets-neural-network-architectures-for","paper_date":"2020-10-21","arxiv_id":"2010.11267","code_links":[{"title":"ARM-software/ML-zoo","url":"https://github.com/ARM-software/ML-zoo"}],"syntology":null}},{"leaderboard":"/sota/keyword-spotting-on-google-speech-commands-v2-4","slug":"keyword-spotting-on-google-speech-commands-v2-4","dataset":"Google Speech Commands (v2)","dataset_url":null,"rows_in_archive":1,"metrics":["Accuracy(10-fold)"],"first_row_in_archive_order":{"model":"Quaternion Neural Networks","paper_title":"Towards on-Device Keyword Spotting using Low-Footprint Quaternion Neural Models","paper_url":"/paper/towards-on-device-keyword-spotting-using-low","paper_date":"2023-09-15","arxiv_id":null,"code_links":[{"title":"DataSenseiAryan/GoogleSpeechCommandLowFootprint","url":"https://github.com/DataSenseiAryan/GoogleSpeechCommandLowFootprint"}],"syntology":null}}],"datasets":[{"url":"/dataset/speech-commands","name":"Speech Commands","full_name":"Speech Commands","num_papers_in_archive":392},{"url":"/dataset/tau-urban-acoustic-scenes-2019","name":"TAU Urban Acoustic Scenes 2019","full_name":"TAU Urban Acoustic Scenes 2019","num_papers_in_archive":14},{"url":"/dataset/voxforge","name":"VoxForge","full_name":"VoxForge","num_papers_in_archive":11},{"url":"/dataset/podcastfillers","name":"PodcastFillers","full_name":"","num_papers_in_archive":5},{"url":"/dataset/google-speech-commands-musan","name":"Google Speech Commands - Musan","full_name":"","num_papers_in_archive":3},{"url":"/dataset/fkd","name":"FKD","full_name":"Football Keywords Dataset","num_papers_in_archive":2},{"url":"/dataset/auto-kws","name":"Auto-KWS","full_name":"","num_papers_in_archive":1},{"url":"/dataset/emospeech","name":"EmoSpeech","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/small-footprint-keyword-spotting","name":"Small-Footprint Keyword Spotting"},{"url":"/task/visual-keyword-spotting","name":"Visual Keyword Spotting"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":113,"tagged_in_all":407,"items":[{"url":"/paper/speech-commands-a-dataset-for-limited","title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","date":"2018-04-09","arxiv_id":"1804.03209","repositories_listed":35,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/hello-edge-keyword-spotting-on","title":"Hello Edge: Keyword Spotting on Microcontrollers","date":"2017-11-20","arxiv_id":"1711.07128","repositories_listed":18,"syntology":null},{"url":"/paper/keyword-transformer-a-self-attention-model","title":"Keyword Transformer: A Self-Attention Model for Keyword Spotting","date":"2021-04-01","arxiv_id":"2104.00769","repositories_listed":10,"syntology":{"n":16,"n_ran":1,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/tera-self-supervised-learning-of-transformer","title":"TERA: Self-Supervised Learning of Transformer Encoder Representation for Speech","date":"2020-07-12","arxiv_id":"2007.06028","repositories_listed":7,"syntology":{"n":14,"n_ran":5,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/ast-audio-spectrogram-transformer","title":"AST: Audio Spectrogram Transformer","date":"2021-04-05","arxiv_id":"2104.01778","repositories_listed":5,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/efficient-keyword-spotting-using-dilated","title":"Efficient keyword spotting using dilated convolutions and gating","date":"2018-11-19","arxiv_id":"1811.07684","repositories_listed":5,"syntology":null},{"url":"/paper/broadcasted-residual-learning-for-efficient","title":"Broadcasted Residual Learning for Efficient Keyword Spotting","date":"2021-06-08","arxiv_id":"2106.04140","repositories_listed":4,"syntology":null},{"url":"/paper/deep-residual-learning-for-small-footprint","title":"Deep Residual Learning for Small-Footprint Keyword Spotting","date":"2017-10-28","arxiv_id":"1710.10361","repositories_listed":4,"syntology":{"n":14,"n_ran":1,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/honk-a-pytorch-reimplementation-of","title":"Honk: A PyTorch Reimplementation of Convolutional Neural Networks for Keyword Spotting","date":"2017-10-18","arxiv_id":"1710.06554","repositories_listed":4,"syntology":null},{"url":"/paper/read-bad-a-new-dataset-and-evaluation-scheme","title":"READ-BAD: A New Dataset and Evaluation Scheme for Baseline Detection in Archival Documents","date":"2017-05-09","arxiv_id":"1705.03311","repositories_listed":4,"syntology":null},{"url":"/paper/ssast-self-supervised-audio-spectrogram","title":"SSAST: Self-Supervised Audio Spectrogram Transformer","date":"2021-10-19","arxiv_id":"2110.09784","repositories_listed":3,"syntology":{"n":16,"n_ran":11,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/temporal-convolution-for-real-time-keyword","title":"Temporal Convolution for Real-time Keyword Spotting on Mobile Devices","date":"2019-04-08","arxiv_id":"1904.03814","repositories_listed":3,"syntology":null},{"url":"/paper/attention-based-end-to-end-models-for-small","title":"Attention-based End-to-End Models for Small-Footprint Keyword Spotting","date":"2018-03-29","arxiv_id":"1803.10916","repositories_listed":3,"syntology":null},{"url":"/paper/an-end-to-end-architecture-for-keyword","title":"An End-to-End Architecture for Keyword Spotting and Voice Activity Detection","date":"2016-11-28","arxiv_id":"1611.09405","repositories_listed":3,"syntology":null},{"url":"/paper/self-learning-for-personalized-keyword","title":"Self-Learning for Personalized Keyword Spotting on Ultra-Low-Power Audio Sensors","date":"2024-08-22","arxiv_id":"2408.12481","repositories_listed":2,"syntology":null},{"url":"/paper/phonmatchnet-phoneme-guided-zero-shot-keyword","title":"PhonMatchNet: Phoneme-Guided Zero-Shot Keyword Spotting for User-Defined Keywords","date":"2023-08-31","arxiv_id":"2308.16511","repositories_listed":2,"syntology":null},{"url":"/paper/improving-label-deficient-keyword-spotting","title":"Improving Label-Deficient Keyword Spotting Through Self-Supervised Pretraining","date":"2022-10-04","arxiv_id":"2210.01703","repositories_listed":2,"syntology":null},{"url":"/paper/paddlespeech-an-easy-to-use-all-in-one-speech-1","title":"PaddleSpeech: An Easy-to-Use All-in-One Speech Toolkit","date":"2022-05-20","arxiv_id":"2205.12007","repositories_listed":2,"syntology":null},{"url":"/paper/progressive-continual-learning-for-spoken","title":"Progressive Continual Learning for Spoken Keyword Spotting","date":"2022-01-29","arxiv_id":"2201.12546","repositories_listed":2,"syntology":null},{"url":"/paper/mlperf-tiny-benchmark","title":"MLPerf Tiny Benchmark","date":"2021-06-14","arxiv_id":"2106.07597","repositories_listed":2,"syntology":{"n":16,"n_ran":1,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/few-shot-keyword-spotting-in-any-language","title":"Few-Shot Keyword Spotting in Any Language","date":"2021-04-03","arxiv_id":"2104.01454","repositories_listed":2,"syntology":null},{"url":"/paper/efficientnet-absolute-zero-for-continuous","title":"EfficientNet-Absolute Zero for Continuous Speech Keyword Spotting","date":"2020-12-31","arxiv_id":"2012.15695","repositories_listed":2,"syntology":null},{"url":"/paper/resource-efficient-dnns-for-keyword-spotting","title":"Resource-efficient DNNs for Keyword Spotting using Neural Architecture Search and Quantization","date":"2020-12-18","arxiv_id":"2012.10138","repositories_listed":2,"syntology":null},{"url":"/paper/decentralizing-feature-extraction-with","title":"Decentralizing Feature Extraction with Quantum Convolutional Neural Network for Automatic Speech Recognition","date":"2020-10-26","arxiv_id":"2010.13309","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/howl-a-deployed-open-source-wake-word","title":"Howl: A Deployed, Open-Source Wake Word Detection System","date":"2020-08-21","arxiv_id":"2008.09606","repositories_listed":2,"syntology":null},{"url":"/paper/federated-learning-for-keyword-spotting","title":"Federated Learning for Keyword Spotting","date":"2018-10-09","arxiv_id":"1810.05512","repositories_listed":2,"syntology":null},{"url":"/paper/trainable-frontend-for-robust-and-far-field","title":"Trainable Frontend For Robust and Far-Field Keyword Spotting","date":"2016-07-19","arxiv_id":"1607.05666","repositories_listed":2,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/glap-general-contrastive-audio-text","title":"GLAP: General contrastive audio-text pretraining across domains and languages","date":"2025-06-12","arxiv_id":"2506.11350","repositories_listed":1,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/advances-in-small-footprint-keyword-spotting","title":"Advances in Small-Footprint Keyword Spotting: A Comprehensive Review of Efficient Models and Algorithms","date":"2025-06-12","arxiv_id":"2506.11169","repositories_listed":1,"syntology":null},{"url":"/paper/chameleon-a-matmul-free-temporal","title":"Chameleon: A MatMul-Free Temporal Convolutional Network Accelerator for End-to-End Few-Shot and Continual Learning from Sequential Data","date":"2025-05-30","arxiv_id":"2505.24852","repositories_listed":1,"syntology":null}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}