{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/24","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":24,"pages_in_order":65,"rows_per_page":100,"rows":[2301,2400],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/23","next":"/task/speech-recognition/papers/25","papers":[{"url":null,"slug":"improving-end-to-end-speech-processing-by","title":"Improving End-to-End Speech Processing by Efficient Text Data Utilization with Latent Synthesis","date":"2023-10-09","arxiv_id":"2310.05374","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-lip-reading-in-romanian-with-cross","title":"End-to-End Lip Reading in Romanian with Cross-Lingual Domain Adaptation and Lateral Inhibition","date":"2023-10-07","arxiv_id":"2310.04858","repositories_listed":0,"syntology":null},{"url":null,"slug":"spike-triggered-contextual-biasing-for-end-to","title":"Spike-Triggered Contextual Biasing for End-to-End Mandarin Speech Recognition","date":"2023-10-07","arxiv_id":"2310.04657","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-privacy-preserving-method-using-secret-key","title":"A privacy-preserving method using secret key for convolutional neural network-based speech classification","date":"2023-10-06","arxiv_id":"2310.04035","repositories_listed":0,"syntology":null},{"url":null,"slug":"hubertopic-enhancing-semantic-representation","title":"HuBERTopic: Enhancing Semantic Representation of HuBERT through Self-supervision Utilizing Topic Model","date":"2023-10-06","arxiv_id":"2310.03975","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-integrated-algorithm-for-robust-and","title":"An Integrated Algorithm for Robust and Imperceptible Audio Adversarial Examples","date":"2023-10-05","arxiv_id":"2310.03349","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-and-insights-exploring-3d-spatial","title":"Challenges and Insights: Exploring 3D Spatial Features and Complex Networks on the MISP Dataset","date":"2023-10-05","arxiv_id":"2310.03901","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-language-model-pruning-for-automatic","title":"Neural Language Model Pruning for Automatic Speech Recognition","date":"2023-10-05","arxiv_id":"2310.03424","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-north-system-for-formosa-speech","title":"The North System for Formosa Speech Recognition Challenge 2023","date":"2023-10-05","arxiv_id":"2310.03443","repositories_listed":0,"syntology":null},{"url":"/paper/universlu-universal-spoken-language","slug":"universlu-universal-spoken-language","title":"UniverSLU: Universal Spoken Language Understanding for Diverse Tasks with Natural Language Instructions","date":"2023-10-04","arxiv_id":"2310.02973","repositories_listed":0,"syntology":null},{"url":null,"slug":"residualtransformer-residual-low-rank","title":"ResidualTransformer: Residual Low-Rank Learning with Weight-Sharing for Transformer Layers","date":"2023-10-03","arxiv_id":"2310.02489","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-model-to-rule-them-all-towards-end-to-end","title":"One model to rule them all ? Towards End-to-End Joint Speaker Diarization and Speech Recognition","date":"2023-10-02","arxiv_id":"2310.01688","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-based-fine-tuning-framework","title":"Active Learning Based Fine-Tuning Framework for Speech Emotion Recognition","date":"2023-09-30","arxiv_id":"2310.00283","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrispeech-200-pan-african-accented-speech","title":"AfriSpeech-200: Pan-African Accented Speech Dataset for Clinical and General Domain ASR","date":"2023-09-30","arxiv_id":"2310.00274","repositories_listed":0,"syntology":null},{"url":null,"slug":"slm-bridge-the-thin-gap-between-speech-and","title":"SLM: Bridge the thin gap between speech and text foundation models","date":"2023-09-30","arxiv_id":"2310.00230","repositories_listed":0,"syntology":null},{"url":null,"slug":"av-cpl-continuous-pseudo-labeling-for-audio","title":"AV-CPL: Continuous Pseudo-Labeling for Audio-Visual Speech Recognition","date":"2023-09-29","arxiv_id":"2309.17395","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-biasing-with-the-knuth-morris","title":"Contextual Biasing with the Knuth-Morris-Pratt Matching Algorithm","date":"2023-09-29","arxiv_id":"2310.00178","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-code-switching-speech-recognition","title":"Enhancing Code-switching Speech Recognition with Interactive Language Biases","date":"2023-09-29","arxiv_id":"2309.16953","repositories_listed":0,"syntology":null},{"url":"/paper/federated-learning-with-differential-privacy","slug":"federated-learning-with-differential-privacy","title":"Enabling Differentially Private Federated Learning for Speech Recognition: Benchmarks, Adaptive Optimizers and Gradient Clipping","date":"2023-09-29","arxiv_id":"2310.00098","repositories_listed":0,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/federated-learning-with-differential-privacy#ran","syntology_url":"https://syntology.ai/paper/2310.00098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00098"}},"official":null}},{"url":null,"slug":"sshr-leveraging-self-supervised-hierarchical","title":"SSHR: Leveraging Self-supervised Hierarchical Representations for Multilingual Automatic Speech Recognition","date":"2023-09-29","arxiv_id":"2309.16937","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-gift-of-feedback-improving-asr-model","title":"The Gift of Feedback: Improving ASR Model Quality by Learning from User Corrections through Federated Learning","date":"2023-09-29","arxiv_id":"2310.00141","repositories_listed":0,"syntology":null},{"url":null,"slug":"wiki-en-asr-adapt-large-scale-synthetic","title":"Wiki-En-ASR-Adapt: Large-scale synthetic dataset for English ASR Customization","date":"2023-09-29","arxiv_id":"2309.17267","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-cross-modality-knowledge","title":"Hierarchical Cross-Modality Knowledge Transfer with Sinkhorn Attention for CTC-based ASR","date":"2023-09-28","arxiv_id":"2309.16093","repositories_listed":0,"syntology":null},{"url":null,"slug":"pp-met-a-real-world-personalized-prompt-based","title":"PP-MeT: a Real-world Personalized Prompt based Meeting Transcription System","date":"2023-09-28","arxiv_id":"2309.16247","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-single-channel-speech-enhancement","title":"Does Single-channel Speech Enhancement Improve Keyword Spotting Accuracy? A Case Study","date":"2023-09-27","arxiv_id":"2309.16060","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speech-recognition-translation-and","title":"Exploring Speech Recognition, Translation, and Understanding with Discrete Speech Units: A Comparative Study","date":"2023-09-27","arxiv_id":"2309.15800","repositories_listed":0,"syntology":null},{"url":"/paper/generative-speech-recognition-error","slug":"generative-speech-recognition-error","title":"Generative Speech Recognition Error Correction with Large Language Models and Task-Activating Prompting","date":"2023-09-27","arxiv_id":"2309.15649","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-adaptation-of-large-language-model","title":"Low-rank Adaptation of Large Language Model Rescoring for Parameter-Efficient Speech Recognition","date":"2023-09-26","arxiv_id":"2309.15223","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-level-vectorized-beam-search-based-on","title":"Segment-Level Vectorized Beam Search Based on Partially Autoregressive Inference","date":"2023-09-26","arxiv_id":"2309.14922","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-pre-training-for-vietnamese-1","title":"Unsupervised Pre-Training for Vietnamese Automatic Speech Recognition in the HYKIST Project","date":"2023-09-26","arxiv_id":"2309.15869","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoprep-an-automatic-preprocessing-framework","title":"AutoPrep: An Automatic Preprocessing Framework for In-the-Wild Speech Data","date":"2023-09-25","arxiv_id":"2309.13905","repositories_listed":0,"syntology":null},{"url":null,"slug":"connecting-speech-encoder-and-large-language","title":"Connecting Speech Encoder and Large Language Model for ASR","date":"2023-09-25","arxiv_id":"2309.13963","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-impact-of-quantization-and-pruning-of","title":"On the Impact of Quantization and Pruning of Self-Supervised Speech Models for Downstream Speech Recognition Tasks \"In-the-Wild''","date":"2023-09-25","arxiv_id":"2309.14462","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-relation-between-internal-language","title":"On the Relation between Internal Language Model and Sequence Discriminative Training for Neural Transducers","date":"2023-09-25","arxiv_id":"2309.14130","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-accent-adaptation-through-masked","title":"Unsupervised Accent Adaptation Through Masked Language Model Correction Of Discrete Self-Supervised Speech Units","date":"2023-09-25","arxiv_id":"2309.13994","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-alignment-with-optimal-transport","title":"Cross-modal Alignment with Optimal Transport for CTC-based ASR","date":"2023-09-24","arxiv_id":"2309.13650","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-with-frequency-domain-auto","title":"Speech enhancement with frequency domain auto-regressive modeling","date":"2023-09-24","arxiv_id":"2309.13537","repositories_listed":0,"syntology":null},{"url":null,"slug":"my-science-tutor-myst-a-large-corpus-of","title":"My Science Tutor (MyST) -- A Large Corpus of Children's Conversational Speech","date":"2023-09-23","arxiv_id":"2309.13347","repositories_listed":0,"syntology":null},{"url":null,"slug":"affect-recognition-in-conversations-using","title":"Affect Recognition in Conversations Using Large Language Models","date":"2023-09-22","arxiv_id":"2309.12881","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-asr-pathways-an-adaptive-masking","title":"Dynamic ASR Pathways: An Adaptive Masking Approach Towards Efficient Pruning of A Multilingual ASR Model","date":"2023-09-22","arxiv_id":"2309.13018","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-of-smoothness-induced-by","title":"Importance of Smoothness Induced by Optimizers in FL4ASR: Towards Understanding Federated Learning for End-to-End ASR","date":"2023-09-22","arxiv_id":"2309.13102","repositories_listed":0,"syntology":null},{"url":null,"slug":"massive-end-to-end-models-for-short-search","title":"Massive End-to-end Models for Short Search Queries","date":"2023-09-22","arxiv_id":"2309.12963","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntt-speaker-diarization-system-for-chime-7","title":"NTT speaker diarization system for CHiME-7: multi-domain, multi-microphone End-to-end and vector clustering diarization","date":"2023-09-22","arxiv_id":"2309.12656","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multiscale-autoencoder-msae-framework-for","title":"A Multiscale Autoencoder (MSAE) Framework for End-to-End Neural Network Speech Enhancement","date":"2023-09-21","arxiv_id":"2309.12121","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsely-shared-lora-on-whisper-for-child","title":"Sparsely Shared LoRA on Whisper for Child Speech Recognition","date":"2023-09-21","arxiv_id":"2309.11756","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-connectionist-temporal-1","title":"Variational Connectionist Temporal Classification for Order-Preserving Sequence Modeling","date":"2023-09-21","arxiv_id":"2309.11983","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiofool-fast-universal-and-synchronization","title":"AudioFool: Fast, Universal and synchronization-free Cross-Domain Attack on Speech Recognition","date":"2023-09-20","arxiv_id":"2309.11462","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-data-collection-and-unsupervised","title":"Leveraging Data Collection and Unsupervised Learning for Code-switched Tunisian Arabic Automatic Speech Recognition","date":"2023-09-20","arxiv_id":"2309.11327","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-audio-representation-as-an","title":"Discrete Audio Representation as an Alternative to Mel-Spectrograms for Speaker and Speech Recognition","date":"2023-09-19","arxiv_id":"2309.10922","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-recognition","title":"End-to-End Speech Recognition Contextualization with Large Language Models","date":"2023-09-19","arxiv_id":"2309.10917","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speech-enhancement-for-low-resource","title":"Exploring Speech Enhancement for Low-resource Speech Synthesis","date":"2023-09-19","arxiv_id":"2309.10795","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-ultrasound-tongue-images-for-1","title":"Incorporating Ultrasound Tongue Images for Audio-Visual Speech Enhancement","date":"2023-09-19","arxiv_id":"2309.10455","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-autoregressive-streaming-asr-with-label","title":"Semi-Autoregressive Streaming ASR With Label Context","date":"2023-09-19","arxiv_id":"2309.10926","repositories_listed":0,"syntology":null},{"url":null,"slug":"cb-whisper-contextual-biasing-whisper-using","title":"A Multitask Training Approach to Enhance Whisper with Contextual Biasing and Open-Vocabulary Keyword Spotting","date":"2023-09-18","arxiv_id":"2309.09552","repositories_listed":0,"syntology":null},{"url":null,"slug":"corpus-synthesis-for-zero-shot-asr-domain","title":"Corpus Synthesis for Zero-shot ASR domain Adaptation using Large Language Models","date":"2023-09-18","arxiv_id":"2309.10707","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-hubert-with-lstms-via-decoupled","title":"Distilling HuBERT with LSTMs via Decoupled Knowledge Distillation","date":"2023-09-18","arxiv_id":"2309.09920","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multilingual-speech-recognition","title":"Enhancing Multilingual Speech Recognition through Language Prompt Tuning and Frame-Level Language Adapter","date":"2023-09-18","arxiv_id":"2309.09443","repositories_listed":0,"syntology":null},{"url":null,"slug":"htec-human-transcription-error-correction","title":"HTEC: Human Transcription Error Correction","date":"2023-09-18","arxiv_id":"2309.10089","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-following-speech-recognition","title":"Instruction-Following Speech Recognition","date":"2023-09-18","arxiv_id":"2309.09843","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-end-to-end-asr-architectures","title":"Investigating End-to-End ASR Architectures for Long Form Audio Transcription","date":"2023-09-18","arxiv_id":"2309.09950","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-modeling-of-the-denoising-process","title":"Continuous Modeling of the Denoising Process for Speech Enhancement Based on Deep Learning","date":"2023-09-17","arxiv_id":"2309.09270","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-end-to-end-multilingual-phoneme","title":"Boosting End-to-End Multilingual Phoneme Recognition through Exploiting Universal Speech Attributes Constraints","date":"2023-09-16","arxiv_id":"2309.08828","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoder-only-architecture-for-speech","title":"Decoder-only Architecture for Speech Recognition with CTC Prompts and Text Data Augmentation","date":"2023-09-16","arxiv_id":"2309.08876","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-for-african","title":"Improving Speech Recognition for African American English With Audio Classification","date":"2023-09-16","arxiv_id":"2309.09996","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-conformers-with-structured-state","title":"Augmenting conformers with structured state-space sequence models for online speech recognition","date":"2023-09-15","arxiv_id":"2309.08551","repositories_listed":0,"syntology":null},{"url":null,"slug":"chunked-attention-based-encoder-decoder-model","title":"Chunked Attention-based Encoder-Decoder Model for Streaming Speech Recognition","date":"2023-09-15","arxiv_id":"2309.08436","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-encoder-supporting-continuous-speech","title":"Combining TF-GridNet and Mixture Encoder for Continuous Speech Separation for Meeting Transcription","date":"2023-09-15","arxiv_id":"2309.08454","repositories_listed":0,"syntology":null},{"url":null,"slug":"t-sot-fnt-streaming-multi-talker-asr-with","title":"t-SOT FNT: Streaming Multi-talker ASR with Text-only Domain Adaptation Capability","date":"2023-09-15","arxiv_id":"2309.08131","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multimodal-information-based-speech-1","title":"The Multimodal Information Based Speech Processing (MISP) 2023 Challenge: Audio-Visual Target Speaker Extraction","date":"2023-09-15","arxiv_id":"2309.08348","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-word-level-end-to-end-neural-speaker","title":"Towards Word-Level End-to-End Neural Speaker Diarization with Auxiliary Network","date":"2023-09-15","arxiv_id":"2309.08489","repositories_listed":0,"syntology":null},{"url":null,"slug":"colld-contrastive-layer-to-layer-distillation","title":"CoLLD: Contrastive Layer-to-layer Distillation for Compressing Multilingual Pre-trained Speech Encoders","date":"2023-09-14","arxiv_id":"2309.07707","repositories_listed":0,"syntology":null},{"url":null,"slug":"cppf-a-contextual-and-post-processing-free","title":"CPPF: A contextual and post-processing-free model for automatic speech recognition","date":"2023-09-14","arxiv_id":"2309.07413","repositories_listed":0,"syntology":null},{"url":null,"slug":"echotune-a-modular-extractor-leveraging-the","title":"Echotune: A Modular Extractor Leveraging the Variable-Length Nature of Speech in ASR Tasks","date":"2023-09-14","arxiv_id":"2309.07765","repositories_listed":0,"syntology":null},{"url":null,"slug":"folding-attention-memory-and-power","title":"Folding Attention: Memory and Power Optimization for On-Device Transformer-based Streaming Speech Recognition","date":"2023-09-14","arxiv_id":"2309.07988","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-attention-based-encoder-decoder-model","title":"Hybrid Attention-based Encoder-decoder Model for Efficient Language Model Adaptation","date":"2023-09-14","arxiv_id":"2309.07369","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-class-based-language-model-for","title":"Incorporating Class-based Language Model for Named Entity Recognition in Factorized Neural Transducer","date":"2023-09-14","arxiv_id":"2309.07648","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxtlm-unified-decoder-only-models-for","title":"Voxtlm: unified decoder-only models for consolidating speech recognition/synthesis and speech/text continuation tasks","date":"2023-09-14","arxiv_id":"2309.07937","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-whisper-perform-speech-based-in-context","title":"Can Whisper perform speech-based in-context learning?","date":"2023-09-13","arxiv_id":"2309.07081","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-child-vocalization-classification","title":"Enhancing Child Vocalization Classification with Phonetically-Tuned Embeddings for Assisting Autism Diagnosis","date":"2023-09-13","arxiv_id":"2309.07287","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vocabulary-keyword-spotting-with","title":"Open-vocabulary Keyword-spotting with Adaptive Instance Normalization","date":"2023-09-13","arxiv_id":"2309.08561","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-learning-synaptic-delays-weights-and","title":"Co-learning synaptic delays, weights and adaptation in spiking neural networks","date":"2023-09-12","arxiv_id":"2311.16112","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-robustness-of-neural-inverse-text","title":"Improving Robustness of Neural Inverse Text Normalization via Data-Augmentation, Semi-Supervised Learning, and Post-Aligning Method","date":"2023-09-12","arxiv_id":"2309.08626","repositories_listed":0,"syntology":null},{"url":null,"slug":"kid-whisper-towards-bridging-the-performance","title":"Kid-Whisper: Towards Bridging the Performance Gap in Automatic Speech Recognition for Children VS. Adults","date":"2023-09-12","arxiv_id":"2309.07927","repositories_listed":0,"syntology":null},{"url":null,"slug":"minuteman-machine-and-human-joining-forces-in","title":"Minuteman: Machine and Human Joining Forces in Meeting Summarization","date":"2023-09-11","arxiv_id":"2309.05272","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-language-models-for","title":"Leveraging Large Language Models for Exploiting ASR Uncertainty","date":"2023-09-09","arxiv_id":"2309.04842","repositories_listed":0,"syntology":null},{"url":null,"slug":"lanser-language-model-supported-speech","title":"LanSER: Language-Model Supported Speech Emotion Recognition","date":"2023-09-07","arxiv_id":"2309.03978","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-representation-transfer-from-large","title":"Multiple Representation Transfer from Large Language Models to End-to-End ASR Systems","date":"2023-09-07","arxiv_id":"2309.04031","repositories_listed":0,"syntology":null},{"url":null,"slug":"rodia-a-new-dataset-for-romanian-dialect","title":"RoDia: A New Dataset for Romanian Dialect Identification from Speech","date":"2023-09-06","arxiv_id":"2309.03378","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-masked-digital-elevation","title":"Self-Supervised Masked Digital Elevation Models Encoding for Low-Resource Downstream Tasks","date":"2023-09-06","arxiv_id":"2309.03367","repositories_listed":0,"syntology":null},{"url":null,"slug":"bring-the-noise-introducing-noise-robustness","title":"Bring the Noise: Introducing Noise Robustness to Pretrained Automatic Speech Recognition","date":"2023-09-05","arxiv_id":"2309.02145","repositories_listed":0,"syntology":null},{"url":null,"slug":"todm-train-once-deploy-many-efficient","title":"TODM: Train Once Deploy Many Efficient Supernet-Based RNN-T Compression For On-device ASR Models","date":"2023-09-05","arxiv_id":"2309.01947","repositories_listed":0,"syntology":null},{"url":null,"slug":"avatar-robust-voice-search-engine-leveraging","title":"AVATAR: Robust Voice Search Engine Leveraging Autoregressive Document Retrieval and Contrastive Learning","date":"2023-09-04","arxiv_id":"2309.01395","repositories_listed":0,"syntology":null},{"url":null,"slug":"sememeasr-boosting-performance-of-end-to-end","title":"SememeASR: Boosting Performance of End-to-End Speech Recognition against Domain and Long-Tailed Data Shift with Sememe Semantic Knowledge","date":"2023-09-04","arxiv_id":"2309.01437","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-only-domain-adaptation-for-end-to-end-1","title":"Text-Only Domain Adaptation for End-to-End Speech Recognition through Down-Sampling Acoustic Representation","date":"2023-09-04","arxiv_id":"2309.02459","repositories_listed":0,"syntology":null},{"url":null,"slug":"mapping-ai-arguments-in-journalism-studies","title":"Mapping AI Arguments in Journalism Studies","date":"2023-09-03","arxiv_id":"2309.12357","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-biasing-of-named-entities-with","title":"Contextual Biasing of Named-Entities with Large Language Models","date":"2023-09-01","arxiv_id":"2309.00723","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpsp-learning-speech-concepts-from-phoneme","title":"Learning Speech Representation From Contrastive Token-Acoustic Pretraining","date":"2023-09-01","arxiv_id":"2309.00424","repositories_listed":0,"syntology":null},{"url":null,"slug":"mi-go-test-framework-which-uses-youtube-as","title":"Mi-Go: Test Framework which uses YouTube as Data Source for Evaluating Speech Recognition Models like OpenAI's Whisper","date":"2023-09-01","arxiv_id":"2309.00329","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-from-non-streaming-to","title":"Knowledge Distillation from Non-streaming to Streaming ASR Encoder using Auxiliary Non-streaming Layer","date":"2023-08-31","arxiv_id":"2308.16415","repositories_listed":0,"syntology":null},{"url":null,"slug":"aster-automatic-speech-recognition-system","title":"ASTER: Automatic Speech Recognition System Accessibility Testing for Stutterers","date":"2023-08-30","arxiv_id":"2308.15742","repositories_listed":0,"syntology":null}],"record_sha256":"582022ff44c63e4b0002184092ede49ac5bae8a93c6d6081ce4297eb83b8c3e0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}