{"url":"/dataset/speech-commands","name":"Speech Commands","full_name":"Speech Commands","description_markdown":"**Speech Commands** is an audio dataset of spoken words designed to help train and evaluate keyword spotting systems .","description_withheld":null,"homepage":"https://arxiv.org/abs/1804.03209","introduced_date":"2018-04-09","introduced_date_note":null,"introduced_by":{"paper":"/paper/speech-commands-a-dataset-for-limited","title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","first_author":"Pete Warden","url":null},"license":{"name":"CC BY","url":null},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Time Series Analysis","url":"/task/time-series","datasets_with_task":"/datasets/task/time-series"},{"name":"Audio Classification","url":"/task/audio-classification","datasets_with_task":"/datasets/task/audio-classification"},{"name":"Federated Learning","url":"/task/federated-learning","datasets_with_task":"/datasets/task/federated-learning"},{"name":"Keyword Spotting","url":"/task/keyword-spotting","datasets_with_task":"/datasets/task/keyword-spotting"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Google Speech Commands","Speech Commands"],"data_loaders":[{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/speech-commands-dataset","frameworks":["tf","pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/speech_commands","frameworks":["tf","jax"]},{"repo":"https://github.com/pytorch/audio","url":"https://pytorch.org/audio/stable/datasets.html#torchaudio.datasets.SPEECHCOMMANDS","frameworks":["pytorch"]},{"repo":"https://github.com/tk-rusch/lem","url":"https://github.com/tk-rusch/lem","frameworks":["pytorch"]}],"num_papers_in_archive":392,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/keyword-spotting-on-google-speech-commands","task":"Keyword Spotting","dataset_variant":"Google Speech Commands","rows":42,"metrics":["Google Speech Commands V1 12","Google Speech Commands V2 12","Google Speech Commands V2 2","Google Speech Commands V2 20","Google Speech Commands V2 35","Google Speech Commands V1 2","Google Speech Commands V1 20","Google Speech Commands V1 35","Google Speech Commands V1 6","10-keyword Speech Commands dataset","Google Speech Command-Musan","% Test Accuracy","Google Speech Commands"],"first_row_in_archive_order":{"model":"TripletLoss-res15","paper":"/paper/learning-efficient-representations-for-3","metrics":{"Google Speech Commands V1 12":"98.56","Google Speech Commands V2 12":"98.37","Google Speech Commands V2 35":"97.0"},"code_links":[{"title":"roman-vygon/triplet_loss_kws","url":"https://github.com/roman-vygon/triplet_loss_kws"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/audio-classification-on-speech-commands-1","task":"Audio Classification","dataset_variant":"Speech Commands","rows":7,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"EAT","paper":"/paper/eat-self-supervised-pre-training-with","metrics":{"Accuracy":"98.3±0.04"},"code_links":[{"title":"cwx-worst-one/eat","url":"https://github.com/cwx-worst-one/eat"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/time-series-on-speech-commands","task":"Time Series Analysis","dataset_variant":"Speech Commands","rows":6,"metrics":["% Test Accuracy","% Test Accuracy (Raw Data)"],"first_row_in_archive_order":{"model":"SepTr","paper":"/paper/septr-separable-transformer-for-audio","metrics":{"% Test Accuracy":"98.51"},"code_links":[{"title":"ristea/septr","url":"https://github.com/ristea/septr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-speech-commands-2","task":"Speech Recognition","dataset_variant":"Speech Commands","rows":3,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Centaurus","paper":"/paper/let-ssms-be-convnets-state-space-modeling","metrics":{"Accuracy (%)":"98.53"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/let-ssms-be-convnets-state-space-modeling","title":"Let SSMs be ConvNets: State-space Modeling with Optimal Tensor Contractions","date":"2025-01-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ssamba-self-supervised-audio-representation","title":"SSAMBA: Self-Supervised Audio Representation Learning with Mamba State Space Model","date":"2024-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/work-in-progress-linear-transformers-for","title":"Work in Progress: Linear Transformers for TinyML","date":"2024-03-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mixer-is-more-than-just-a-model","title":"Mixer is more than just a model","date":"2024-02-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/eat-self-supervised-pre-training-with","title":"EAT: Self-Supervised Pre-Training with Efficient Audio Transformer","date":"2024-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":13,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-on-device-keyword-spotting-using-low","title":"Towards on-Device Keyword Spotting using Low-Footprint Quaternion Neural Models","date":"2023-09-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/masked-modeling-duo-learning-representations","title":"Masked Modeling Duo: Learning Representations by Encouraging Both Networks to Model the Input","date":"2022-10-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/liquid-structural-state-space-models","title":"Liquid Structural State-Space Models","date":"2022-09-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficientleaf-a-faster-learnable-audio","title":"EfficientLEAF: A Faster LEarnable Audio Frontend of Questionable Use","date":"2022-07-12","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-audio-strikes-back-boosting","title":"End-to-End Audio Strikes Back: Boosting Augmentations Towards An Efficient Audio Classification Network","date":"2022-04-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/septr-separable-transformer-for-audio","title":"SepTr: Separable Transformer for Audio Spectrogram Processing","date":"2022-03-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hts-at-a-hierarchical-token-semantic-audio","title":"HTS-AT: A Hierarchical Token-Semantic Audio Transformer for Sound Classification and Detection","date":"2022-02-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/importantaug-a-data-augmentation-agent-for","title":"ImportantAug: a data augmentation agent for speech","date":"2021-12-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficiently-modeling-long-sequences-with-1","title":"Efficiently Modeling Long Sequences with Structured State Spaces","date":"2021-10-31","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":55,"samples_ran":28,"samples_unverified":27,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flexconv-continuous-kernel-convolutions-with-1","title":"FlexConv: Continuous Kernel Convolutions with Differentiable Kernel Sizes","date":"2021-10-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/attention-free-keyword-spotting","title":"Attention-Free Keyword Spotting","date":"2021-10-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":0,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/broadcasted-residual-learning-for-efficient","title":"Broadcasted Residual Learning for Efficient Keyword Spotting","date":"2021-06-08","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/wav2kws-transfer-learning-from-speech","title":"Wav2KWS: Transfer Learning from Speech Representations for Keyword Spotting","date":"2021-05-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/polynomial-networks-in-deep-classifiers","title":"Augmenting Deep Classifiers with Polynomial Neural Networks","date":"2021-04-16","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-keyword-spotting-using-neural","title":"End-to-end Keyword Spotting using Neural Architecture Search and Quantization","date":"2021-04-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ast-audio-spectrogram-transformer","title":"AST: Audio Spectrogram Transformer","date":"2021-04-05","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pate-aae-incorporating-adversarial","title":"PATE-AAE: Incorporating Adversarial Autoencoder into Private Aggregation of Teacher Ensembles for Spoken Command Classification","date":"2021-04-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/keyword-transformer-a-self-attention-model","title":"Keyword Transformer: A Self-Attention Model for Keyword Spotting","date":"2021-04-01","rows_on_this_dataset":3,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":1,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/subspectral-normalization-for-neural-audio","title":"SubSpectral Normalization for Neural Audio Data Processing","date":"2021-03-25","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/edgecrnn-an-edgecomputing-oriented-model-of","title":"EdgeCRNN: an edgecomputing oriented model of acoustic feature enhancement for keyword spotting","date":"2021-03-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ckconv-continuous-kernel-convolution-for","title":"CKConv: Continuous Kernel Convolution For Sequential Data","date":"2021-02-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-efficient-representations-for-3","title":"Learning Efficient Representations for Keyword Spotting with Triplet Loss","date":"2021-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/decentralizing-feature-extraction-with","title":"Decentralizing Feature Extraction with Quantum Convolutional Neural Network for Automatic Speech Recognition","date":"2020-10-26","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/micronets-neural-network-architectures-for","title":"MicroNets: Neural Network Architectures for Deploying TinyML Applications on Commodity Microcontrollers","date":"2020-10-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/neural-architecture-search-for-keyword","title":"Neural Architecture Search For Keyword Spotting","date":"2020-09-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/howl-a-deployed-open-source-wake-word","title":"Howl: A Deployed, Open-Source Wake Word Detection System","date":"2020-08-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/matchboxnet-1d-time-channel-separable-1","title":"MatchboxNet: 1D Time-Channel Separable Convolutional Neural Network Architecture for Speech Commands Recognition","date":"2020-04-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/training-keyword-spotters-with-limited-and","title":"Training Keyword Spotters with Limited and Synthesized Speech Data","date":"2020-01-31","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multi-layer-attention-mechanism-for-speech","title":"Multi-layer Attention Mechanism for Speech Keyword Recognition","date":"2019-07-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/temporal-convolution-for-real-time-keyword","title":"Temporal Convolution for Real-time Keyword Spotting on Mobile Devices","date":"2019-04-08","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/effective-combination-of-densenet-andbilstm","title":"Effective Combination of DenseNet andBiLSTM for Keyword Spotting","date":"2019-01-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-keyword-spotting-using-time-delay","title":"Efficient keyword spotting using time delay neural networks","date":"2018-08-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/a-neural-attention-model-for-speech-command","title":"A neural attention model for speech command recognition","date":"2018-08-27","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hello-edge-keyword-spotting-on","title":"Hello Edge: Keyword Spotting on Microcontrollers","date":"2017-11-20","rows_on_this_dataset":6,"code_links":18,"syntology":null},{"paper":"/paper/streaming-keyword-spotting-on-mobile-devices","title":"Streaming keyword spotting on mobile devices","date":null,"rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/convmixer-feature-interactive-convolution","title":"ConvMixer: Feature Interactive Convolution with Curriculum Learning for Small Footprint and Noisy Far-field Keyword Spotting","date":null,"rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":12,"samples_harvested":133,"samples_ran":61,"samples_unverified":72,"pointer_only_for_licence":16,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}