{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/keyword-spotting/papers/2","list_of":"/task/keyword-spotting","task":"Keyword Spotting","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":407,"counts":{"archive_papers_tagged":407,"with_a_code_link":113,"where_syntology_ran_a_sample":18,"not_listed_spam_title":0,"listed":407,"listed_where_code_ran":18,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":17,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":17,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/keyword-spotting","prev":"/task/keyword-spotting","next":"/task/keyword-spotting/papers/3","papers":[{"url":"/paper/temporal-feedback-convolutional-recurrent","slug":"temporal-feedback-convolutional-recurrent","title":"Temporal Feedback Convolutional Recurrent Neural Networks for Speech Command Recognition","date":"2019-10-30","arxiv_id":"1911.01803","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-example-detection-by","slug":"adversarial-example-detection-by","title":"Adversarial Example Detection by Classification for Deep Speech Recognition","date":"2019-10-22","arxiv_id":"1910.10013","repositories_listed":1,"syntology":null},{"url":"/paper/indian-emospeech-command-dataset-a-dataset","slug":"indian-emospeech-command-dataset-a-dataset","title":"Indian EmoSpeech Command Dataset: A dataset for emotion based speech recognition in the wild","date":"2019-10-18","arxiv_id":"1910.13801","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-sequence-to-sequence-models-for","slug":"evaluating-sequence-to-sequence-models-for","title":"Evaluating Sequence-to-Sequence Models for Handwritten Text Recognition","date":"2019-03-18","arxiv_id":"1903.07377","repositories_listed":1,"syntology":null},{"url":"/paper/meta-learning-for-few-shot-keyword-spotting","slug":"meta-learning-for-few-shot-keyword-spotting","title":"An Investigation of Few-Shot Learning in Spoken Term Classification","date":"2018-12-26","arxiv_id":"1812.10233","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-keyword-spotting-efficiency-on","slug":"benchmarking-keyword-spotting-efficiency-on","title":"Benchmarking Keyword Spotting Efficiency on Neuromorphic Hardware","date":"2018-12-04","arxiv_id":"1812.01739","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-keyword-spotting-efficiency-on#ran","syntology_url":"https://syntology.ai/paper/1812.01739","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.01739"}},"official":{"repos":["abr/power_benchmarks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/donut-ctc-based-query-by-example-keyword","slug":"donut-ctc-based-query-by-example-keyword","title":"DONUT: CTC-based Query-by-Example Keyword Spotting","date":"2018-11-26","arxiv_id":"1811.10736","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-adaptive-neural-architecture","slug":"stochastic-adaptive-neural-architecture","title":"Stochastic Adaptive Neural Architecture Search for Keyword Spotting","date":"2018-11-16","arxiv_id":"1811.06753","repositories_listed":1,"syntology":null},{"url":"/paper/javascript-convolutional-neural-networks-for","slug":"javascript-convolutional-neural-networks-for","title":"JavaScript Convolutional Neural Networks for Keyword Spotting in the Browser: An Experimental Analysis","date":"2018-10-30","arxiv_id":"1810.12859","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-keyword-spotting-for-visual-speech","slug":"zero-shot-keyword-spotting-for-visual-speech","title":"Zero-shot keyword spotting for visual speech recognition in-the-wild","date":"2018-07-23","arxiv_id":"1807.08469","repositories_listed":1,"syntology":null},{"url":"/paper/gtm-uvigo-systems-for-the-query-by-example","slug":"gtm-uvigo-systems-for-the-query-by-example","title":"GTM-UVigo Systems for the Query-by-Example Search on Speech Task at MediaEval 2015","date":"2015-09-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/whats-cookin-interpreting-cooking-videos","slug":"whats-cookin-interpreting-cooking-videos","title":"What’s Cookin’? Interpreting Cooking Videos using Text, Speech and Vision","date":"2015-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/whats-cookin-interpreting-cooking-videos-1","slug":"whats-cookin-interpreting-cooking-videos-1","title":"What's Cookin'? Interpreting Cooking Videos using Text, Speech and Vision","date":"2015-03-05","arxiv_id":"1503.01558","repositories_listed":1,"syntology":null},{"url":null,"slug":"enhancing-few-shot-keyword-spotting","title":"Enhancing Few-shot Keyword Spotting Performance through Pre-Trained Self-supervised Speech Models","date":"2025-06-21","arxiv_id":"2506.17686","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-keyword-spotting-using","title":"Low-resource keyword spotting using contrastively trained transformer acoustic word embeddings","date":"2025-06-21","arxiv_id":"2506.17690","repositories_listed":0,"syntology":null},{"url":null,"slug":"asap-fe-energy-efficient-feature-extraction","title":"ASAP-FE: Energy-Efficient Feature Extraction Enabling Multi-Channel Keyword Spotting on Edge Processors","date":"2025-06-17","arxiv_id":"2506.14657","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08346","title":"SPBA: Utilizing Speech Large Language Model for Backdoor Attacks on Speech Classification Models","date":"2025-06-10","arxiv_id":"2506.08346","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08911","title":"Implementing Keyword Spotting on the MCUX947 Microcontroller with Integrated NPU","date":"2025-06-10","arxiv_id":"2506.08911","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-impact-of-anisotropy-in-neural","title":"Assessing the Impact of Anisotropy in Neural Representations of Speech: A Case Study on Keyword Spotting","date":"2025-06-06","arxiv_id":"2506.11096","repositories_listed":0,"syntology":null},{"url":null,"slug":"wctc-biasing-retraining-free-contextual","title":"WCTC-Biasing: Retraining-free Contextual Biasing ASR with Wildcard CTC-based Keyword Spotting and Inter-layer Biasing","date":"2025-06-02","arxiv_id":"2506.01263","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-unlearning","title":"Speech Unlearning","date":"2025-06-01","arxiv_id":"2506.00848","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-synth4kws-scalable-automatic-generation","title":"LLM-Synth4KWS: Scalable Automatic Generation and Synthesis of Confusable Data for Custom Keyword Spotting","date":"2025-05-29","arxiv_id":"2505.22995","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-deep-metric-learning-for-cross","title":"Adversarial Deep Metric Learning for Cross-Modal Audio-Text Alignment in Open-Vocabulary Keyword Spotting","date":"2025-05-22","arxiv_id":"2505.16735","repositories_listed":0,"syntology":null},{"url":null,"slug":"adakws-towards-robust-keyword-spotting-with","title":"AdaKWS: Towards Robust Keyword Spotting with Test-Time Adaptation","date":"2025-05-20","arxiv_id":"2505.14600","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphemeaug-a-systematic-approach-to","title":"GraphemeAug: A Systematic Approach to Synthesized Hard Negative Keyword Spotting Examples","date":"2025-05-20","arxiv_id":"2505.14814","repositories_listed":0,"syntology":null},{"url":null,"slug":"analytickws-towards-exemplar-free-analytic","title":"AnalyticKWS: Towards Exemplar-Free Analytic Class Incremental Learning for Small-footprint Keyword Spotting","date":"2025-05-17","arxiv_id":"2505.11817","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-noise-resilient-keyword-spotting","title":"Adaptive Noise Resilient Keyword Spotting Using One-Shot Learning","date":"2025-05-14","arxiv_id":"2505.09304","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-continual-learning-in-keyword","title":"Efficient Continual Learning in Keyword Spotting using Binary Neural Networks","date":"2025-05-05","arxiv_id":"2505.02469","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardware-software-co-design-of-risc-v","title":"Hardware/Software Co-Design of RISC-V Extensions for Accelerating Sparse DNNs on FPGAs","date":"2025-04-28","arxiv_id":"2504.19659","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-powered-agile-analog-circuit-design-and","title":"AI-Powered Agile Analog Circuit Design and Optimization","date":"2025-04-17","arxiv_id":"2505.03750","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-keyword-spotting-using","title":"Towards efficient keyword spotting using spike-based time difference encoders","date":"2025-03-19","arxiv_id":"2503.15402","repositories_listed":0,"syntology":null},{"url":null,"slug":"eventprop-training-for-efficient-neuromorphic","title":"Eventprop training for efficient neuromorphic applications","date":"2025-03-06","arxiv_id":"2503.04341","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-noise-robust-whisper-keyword-spotting","title":"Toward noise-robust whisper keyword spotting on headphones with in-earcup microphone and curriculum learning","date":"2025-02-01","arxiv_id":"2502.00295","repositories_listed":0,"syntology":null},{"url":"/paper/let-ssms-be-convnets-state-space-modeling","slug":"let-ssms-be-convnets-state-space-modeling","title":"Let SSMs be ConvNets: State-space Modeling with Optimal Tensor Contractions","date":"2025-01-22","arxiv_id":"2501.13230","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-agnostic-multitask-whisper-training-for","title":"Noise-Agnostic Multitask Whisper Training for Reducing False Alarm Errors in Call-for-Help Detection","date":"2025-01-20","arxiv_id":"2501.11631","repositories_listed":0,"syntology":null},{"url":null,"slug":"vocal-tract-length-warped-features-for-spoken","title":"Vocal Tract Length Warped Features for Spoken Keyword Spotting","date":"2025-01-07","arxiv_id":"2501.03523","repositories_listed":0,"syntology":null},{"url":null,"slug":"phoneme-level-contrastive-learning-for-user","title":"Phoneme-Level Contrastive Learning for User-Defined Keyword Spotting with Flexible Enrollment","date":"2024-12-30","arxiv_id":"2412.20805","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-aware-adapter-for-few-shot-keyword","title":"Text-Aware Adapter for Few-Shot Keyword Spotting","date":"2024-12-24","arxiv_id":"2412.18142","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntc-kws-noise-aware-ctc-for-robust-keyword","title":"NTC-KWS: Noise-aware CTC for Robust Keyword Spotting","date":"2024-12-17","arxiv_id":"2412.12614","repositories_listed":0,"syntology":null},{"url":null,"slug":"ghostrnn-reducing-state-redundancy-in-rnn","title":"GhostRNN: Reducing State Redundancy in RNN with Cheap Operations","date":"2024-11-20","arxiv_id":"2411.14489","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-temporal-resolution-domain","title":"Zero-Shot Temporal Resolution Domain Adaptation for Spiking Neural Networks","date":"2024-11-07","arxiv_id":"2411.04760","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-hearing-aid-voice-control","title":"Noise-Robust Hearing Aid Voice Control","date":"2024-11-05","arxiv_id":"2411.03150","repositories_listed":0,"syntology":null},{"url":null,"slug":"ge2e-kws-generalized-end-to-end-training-and","title":"GE2E-KWS: Generalized End-to-End Training and Evaluation for Zero-shot Keyword Spotting","date":"2024-10-22","arxiv_id":"2410.16647","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-literature-review-of-keyword-spotting","title":"A Literature Review of Keyword Spotting Technologies for Urdu","date":"2024-09-16","arxiv_id":"2409.16317","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-integration-of-kan-for-keyword","title":"Effective Integration of KAN for Keyword Spotting","date":"2024-09-13","arxiv_id":"2409.08605","repositories_listed":0,"syntology":null},{"url":null,"slug":"dark-experience-for-incremental-keyword","title":"Dark Experience for Incremental Keyword Spotting","date":"2024-09-12","arxiv_id":"2409.08153","repositories_listed":0,"syntology":null},{"url":null,"slug":"slick-exploiting-subsequences-for-length","title":"SLiCK: Exploiting Subsequences for Length-Constrained Keyword Spotting","date":"2024-09-06","arxiv_id":"2409.09067","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-laryngoscopic-video-analysis-for","title":"Multimodal Laryngoscopic Video Analysis for Assisted Diagnosis of Vocal Fold Paralysis","date":"2024-09-05","arxiv_id":"2409.03597","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-augmentation-an-unsupervised","title":"Contrastive Augmentation: An Unsupervised Learning Approach for Keyword Spotting in Speech Technology","date":"2024-08-31","arxiv_id":"2409.00356","repositories_listed":0,"syntology":null},{"url":null,"slug":"emoattack-utilizing-emotional-voice","title":"EmoAttack: Utilizing Emotional Voice Conversion for Speech Backdoor Attacks on Deep Speech Classification Models","date":"2024-08-28","arxiv_id":"2408.15508","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-by-example-keyword-spotting-using","title":"Query-by-Example Keyword Spotting Using Spectral-Temporal Graph Attentive Pooling and Multi-Task Learning","date":"2024-08-27","arxiv_id":"2409.00099","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-training-with-adversarial","title":"Disentangled Training with Adversarial Examples For Robust Small-footprint Keyword Spotting","date":"2024-08-23","arxiv_id":"2408.13355","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-training-of-keyword-spotting-to","title":"Adversarial training of Keyword Spotting to Minimize TTS Data Overfitting","date":"2024-08-20","arxiv_id":"2408.10463","repositories_listed":0,"syntology":null},{"url":null,"slug":"convexity-based-pruning-of-speech","title":"Convexity-based Pruning of Speech Representation Models","date":"2024-08-16","arxiv_id":"2408.11858","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-audio-and-text-using","title":"Bridging the Gap between Audio and Text using Parallel-attention for User-defined Keyword Spotting","date":"2024-08-07","arxiv_id":"2408.03593","repositories_listed":0,"syntology":null},{"url":null,"slug":"frequency-channel-attention-network-for-small","title":"Frequency & Channel Attention Network for Small Footprint Noisy Spoken Keyword Spotting","date":"2024-07-29","arxiv_id":"2407.19834","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-temporal-hierarchy-in-spiking","title":"The Role of Temporal Hierarchy in Spiking Neural Networks","date":"2024-07-26","arxiv_id":"2407.18838","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-tts-synthesized-data-for-efficient","title":"Utilizing TTS Synthesized Data for Efficient Development of Keyword Spotting Model","date":"2024-07-26","arxiv_id":"2407.18879","repositories_listed":0,"syntology":null},{"url":null,"slug":"synth4kws-synthesized-speech-for-user-defined","title":"Synth4Kws: Synthesized Speech for User Defined Keyword Spotting in Low Resource Environments","date":"2024-07-23","arxiv_id":"2407.16840","repositories_listed":0,"syntology":null},{"url":null,"slug":"kwt-tiny-risc-v-accelerated-embedded-keyword","title":"KWT-Tiny: RISC-V Accelerated, Embedded Keyword Spotting Transformer","date":"2024-07-22","arxiv_id":"2407.16026","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-boundaries-of-on-device","title":"Exploring the Boundaries of On-Device Inference: When Tiny Falls Short, Go Hierarchical","date":"2024-07-10","arxiv_id":"2407.11061","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitaper-mel-spectrograms-for-keyword","title":"Multitaper mel-spectrograms for keyword spotting","date":"2024-07-05","arxiv_id":"2407.04662","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-airport-tower-command-recognition","title":"Advancing Airport Tower Command Recognition: Integrating Squeeze-and-Excitation and Broadcasted Residual Learning","date":"2024-06-26","arxiv_id":"2406.18313","repositories_listed":0,"syntology":null},{"url":null,"slug":"micro-power-spoken-keyword-spotting-on-xylo","title":"Micro-power spoken keyword spotting on Xylo Audio 2","date":"2024-06-21","arxiv_id":"2406.15112","repositories_listed":0,"syntology":null},{"url":null,"slug":"dasb-discrete-audio-and-speech-benchmark","title":"DASB -- Discrete Audio and Speech Benchmark","date":"2024-06-20","arxiv_id":"2406.14294","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-local-convolution-with-spiking-neural","title":"Global-Local Convolution with Spiking Neural Networks for Energy-efficient Keyword Spotting","date":"2024-06-19","arxiv_id":"2406.13179","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-aligned-audio-text-embedding-for","title":"CTC-aligned Audio-Text Embedding for Streaming Open-vocabulary Keyword Spotting","date":"2024-06-12","arxiv_id":"2406.07923","repositories_listed":0,"syntology":null},{"url":null,"slug":"relational-proxy-loss-for-audio-text-based","title":"Relational Proxy Loss for Audio-Text based Keyword Spotting","date":"2024-06-08","arxiv_id":"2406.05314","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-guided-adaptation-of-automatic-speech","title":"Keyword-Guided Adaptation of Automatic Speech Recognition","date":"2024-06-04","arxiv_id":"2406.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"repcnn-micro-sized-mighty-models-for-wakeword","title":"RepCNN: Micro-sized, Mighty Models for Wakeword Detection","date":"2024-06-04","arxiv_id":"2406.02652","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinysv-speaker-verification-in-tinyml-with-on","title":"TinySV: Speaker Verification in TinyML with On-device Learning","date":"2024-06-03","arxiv_id":"2406.01655","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-user-defined-keyword-spotting","title":"End-to-End User-Defined Keyword Spotting using Shifted Delta Coefficients","date":"2024-05-23","arxiv_id":"2405.14489","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-contactless-elevators-with-tinyml","title":"Towards Contactless Elevators with TinyML using CNN-based Person Detection and Keyword Spotting","date":"2024-05-19","arxiv_id":"2405.13051","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-65nm-36nj-decision-bio-inspired-temporal","title":"DeltaKWS: A 65nm 36nJ/Decision Bio-inspired Temporal-Sparsity-Aware Digital Keyword Spotting IC with 0.6V Near-Threshold SRAM","date":"2024-05-06","arxiv_id":"2405.03905","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-sample-dynamic-time-warping-for-few","title":"Multi-Sample Dynamic Time Warping for Few-Shot Keyword Spotting","date":"2024-04-23","arxiv_id":"2404.14903","repositories_listed":0,"syntology":null},{"url":"/paper/work-in-progress-linear-transformers-for","slug":"work-in-progress-linear-transformers-for","title":"Work in Progress: Linear Transformers for TinyML","date":"2024-03-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"more-than-words-advancements-and-challenges","title":"More than words: Advancements and challenges in speech recognition for singing","date":"2024-03-14","arxiv_id":"2403.09298","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-domain-learning-for-keyword","title":"On-Device Domain Learning for Keyword Spotting on Low-Power Extreme Edge Embedded Systems","date":"2024-03-12","arxiv_id":"2403.10549","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-wav2vec2-embeddings-for-on","title":"A Closer Look at Wav2Vec2 Embeddings for On-Device Single-Channel Speech Enhancement","date":"2024-03-03","arxiv_id":"2403.01369","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-acoustic-word-embeddings-for","title":"Multilingual acoustic word embeddings for zero-resource languages","date":"2024-01-19","arxiv_id":"2401.10543","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-learning-with-audio","title":"Contrastive Learning With Audio Discrimination For Customizable Keyword Spotting In Continuous Speech","date":"2024-01-12","arxiv_id":"2401.06485","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-adversarial-audio","title":"Maximum-Entropy Adversarial Audio Augmentation for Keyword Spotting","date":"2024-01-12","arxiv_id":"2401.06897","repositories_listed":0,"syntology":null},{"url":null,"slug":"u2-kws-unified-two-pass-open-vocabulary","title":"U2-KWS: Unified Two-pass Open-vocabulary Keyword Spotting with Keyword Bias","date":"2023-12-15","arxiv_id":"2312.09760","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-spotting-detecting-commands-in-speech","title":"Keyword spotting -- Detecting commands in speech using deep learning","date":"2023-12-09","arxiv_id":"2312.05640","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalizing-keyword-spotting-with-speaker","title":"Personalizing Keyword Spotting with Speaker Information","date":"2023-11-06","arxiv_id":"2311.03419","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-single-channel-speech-enhancement","title":"Does Single-channel Speech Enhancement Improve Keyword Spotting Accuracy? A Case Study","date":"2023-09-27","arxiv_id":"2309.16060","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-non-associativity-of-analog","title":"On the Non-Associativity of Analog Computations","date":"2023-09-25","arxiv_id":"2309.14292","repositories_listed":0,"syntology":null},{"url":null,"slug":"vic-kd-variance-invariance-covariance","title":"VIC-KD: Variance-Invariance-Covariance Knowledge Distillation to Make Keyword Spotting More Robust Against Adversarial Attacks","date":"2023-09-22","arxiv_id":"2309.12914","repositories_listed":0,"syntology":null},{"url":null,"slug":"cb-whisper-contextual-biasing-whisper-using","title":"A Multitask Training Approach to Enhance Whisper with Contextual Biasing and Open-Vocabulary Keyword Spotting","date":"2023-09-18","arxiv_id":"2309.09552","repositories_listed":0,"syntology":null},{"url":null,"slug":"spiking-leaf-a-learnable-auditory-front-end","title":"Spiking-LEAF: A Learnable Auditory front-end for Spiking Neural Networks","date":"2023-09-18","arxiv_id":"2309.09469","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-keyword-spotting-with","title":"Open-vocabulary Keyword-spotting with Adaptive Instance Normalization","date":"2023-09-13","arxiv_id":"2309.08561","repositories_listed":0,"syntology":null},{"url":null,"slug":"iphonmatchnet-zero-shot-user-defined-keyword","title":"iPhonMatchNet: Zero-Shot User-Defined Keyword Spotting Using Implicit Acoustic Echo Cancellation","date":"2023-09-12","arxiv_id":"2309.06096","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-language-models-for","title":"Leveraging Large Language Models for Exploiting ASR Uncertainty","date":"2023-09-09","arxiv_id":"2309.04842","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-self-supervised-learning-of","title":"Understanding Self-Supervised Learning of Speech Representation via Invariance and Redundancy Reduction","date":"2023-09-07","arxiv_id":"2309.03619","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-small-footprint-few-shot-keyword","title":"Improving Small Footprint Few-shot Keyword Spotting with Supervision on Auxiliary Data","date":"2023-08-31","arxiv_id":"2309.00647","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-vision-inspired-keyword-spotting","title":"Improving vision-inspired keyword spotting using dynamic module skipping in streaming conformer encoder","date":"2023-08-31","arxiv_id":"2309.00140","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-keyword-spotting-based-on","title":"Flexible Keyword Spotting based on Homogeneous Audio-Text Embedding","date":"2023-08-12","arxiv_id":"2308.06472","repositories_listed":0,"syntology":null},{"url":null,"slug":"custom-dnn-using-reward-modulated-inverted","title":"Custom DNN using Reward Modulated Inverted STDP Learning for Temporal Pattern Recognition","date":"2023-07-15","arxiv_id":"2307.07869","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-constrained-self-supervised-speech","title":"On-Device Constrained Self-Supervised Speech Representation Learning for Keyword Spotting via Knowledge Distillation","date":"2023-07-06","arxiv_id":"2307.02720","repositories_listed":0,"syntology":null},{"url":null,"slug":"matching-latent-encoding-for-audio-text-based","title":"Matching Latent Encoding for Audio-Text based Keyword Spotting","date":"2023-06-08","arxiv_id":"2306.05245","repositories_listed":0,"syntology":null}],"record_sha256":"0eca6e08cee852086dac19646ef95cbe807bcea40c28fb645e35d25f678962fd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}