{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/7","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":31,"rows_per_page":100,"rows":[601,700],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/6","next":"/task/automatic-speech-recognition/papers/8","papers":[{"url":"/paper/lstm-benchmarks-for-deep-learning-frameworks","slug":"lstm-benchmarks-for-deep-learning-frameworks","title":"LSTM Benchmarks for Deep Learning Frameworks","date":"2018-06-05","arxiv_id":"1806.01818","repositories_listed":1,"syntology":null},{"url":"/paper/targeted-adversarial-examples-for-black-box","slug":"targeted-adversarial-examples-for-black-box","title":"Targeted Adversarial Examples for Black Box Audio Systems","date":"2018-05-20","arxiv_id":"1805.07820","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/targeted-adversarial-examples-for-black-box#ran","syntology_url":"https://syntology.ai/paper/1805.07820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07820"}},"official":{"repos":["rtaori/Black-Box-Audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/interpersonal-relationship-labels-for-the","slug":"interpersonal-relationship-labels-for-the","title":"Interpersonal Relationship Labels for the CALLHOME Corpus","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/syllable-based-sequence-to-sequence-speech","slug":"syllable-based-sequence-to-sequence-speech","title":"Syllable-Based Sequence-to-Sequence Speech Recognition with the Transformer in Mandarin Chinese","date":"2018-04-28","arxiv_id":"1804.10752","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-sequence-to-sequence-learning-for","slug":"attentive-sequence-to-sequence-learning-for","title":"Attentive Sequence-to-Sequence Learning for Diacritic Restoration of Yorùbá Language Text","date":"2018-04-03","arxiv_id":"1804.00832","repositories_listed":1,"syntology":null},{"url":"/paper/light-gated-recurrent-units-for-speech","slug":"light-gated-recurrent-units-for-speech","title":"Light Gated Recurrent Units for Speech Recognition","date":"2018-03-26","arxiv_id":"1803.10225","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-librispeech-with-french","slug":"augmenting-librispeech-with-french","title":"Augmenting Librispeech with French Translations: A Multimodal Corpus for Direct Speech Translation Evaluation","date":"2018-02-09","arxiv_id":"1802.03142","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-past-mistakes-improving","slug":"learning-from-past-mistakes-improving","title":"Learning from Past Mistakes: Improving Automatic Speech Recognition Output via Noisy-Clean Phrase Context Modeling","date":"2018-02-07","arxiv_id":"1802.02607","repositories_listed":1,"syntology":null},{"url":"/paper/did-you-hear-that-adversarial-examples","slug":"did-you-hear-that-adversarial-examples","title":"Did you hear that? Adversarial Examples Against Automatic Speech Recognition","date":"2018-01-02","arxiv_id":"1801.00554","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-hidden-representations-in-end-to","slug":"analyzing-hidden-representations-in-end-to","title":"Analyzing Hidden Representations in End-to-End Automatic Speech Recognition Systems","date":"2017-09-13","arxiv_id":"1709.04482","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-english-intelligibility-remediation","slug":"spoken-english-intelligibility-remediation","title":"Spoken English Intelligibility Remediation with PocketSphinx Alignment and Feature Extraction Improves Substantially over the State of the Art","date":"2017-09-06","arxiv_id":"1709.01713","repositories_listed":1,"syntology":null},{"url":"/paper/language-identification-using-deep","slug":"language-identification-using-deep","title":"Language Identification Using Deep Convolutional Recurrent Neural Networks","date":"2017-08-16","arxiv_id":"1708.04811","repositories_listed":1,"syntology":null},{"url":"/paper/massively-multilingual-neural-grapheme-to","slug":"massively-multilingual-neural-grapheme-to","title":"Massively Multilingual Neural Grapheme-to-Phoneme Conversion","date":"2017-08-04","arxiv_id":"1708.01464","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-submodular-rank-aggregation-on","slug":"unsupervised-submodular-rank-aggregation-on","title":"Unsupervised Submodular Rank Aggregation on Score-based Permutations","date":"2017-07-04","arxiv_id":"1707.01166","repositories_listed":1,"syntology":null},{"url":"/paper/improving-lstm-ctc-based-asr-performance-in","slug":"improving-lstm-ctc-based-asr-performance-in","title":"Improving LSTM-CTC based ASR performance in domains with limited training data","date":"2017-07-03","arxiv_id":"1707.00722","repositories_listed":1,"syntology":null},{"url":"/paper/joint-ctcattention-decoding-for-end-to-end","slug":"joint-ctcattention-decoding-for-end-to-end","title":"Joint CTC/attention decoding for end-to-end speech recognition","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/speech-based-visual-question-answering","slug":"speech-based-visual-question-answering","title":"Speech-Based Visual Question Answering","date":"2017-05-01","arxiv_id":"1705.00464","repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-speech-recognition-with","slug":"towards-end-to-end-speech-recognition-with","title":"Towards End-to-End Speech Recognition with Deep Convolutional Neural Networks","date":"2017-01-10","arxiv_id":"1701.02720","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/towards-end-to-end-speech-recognition-with#ran","syntology_url":"https://syntology.ai/paper/1701.02720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.02720"}},"official":null}},{"url":"/paper/audio-segmentation-for-robust-real-time","slug":"audio-segmentation-for-robust-real-time","title":"Audio Segmentation for Robust Real-Time Speech Recognition Based on Neural Networks","date":"2016-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/latent-tree-language-model-1","slug":"latent-tree-language-model-1","title":"Latent Tree Language Model","date":"2016-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/collecting-resources-in-sub-saharan-african","slug":"collecting-resources-in-sub-saharan-african","title":"Collecting Resources in Sub-Saharan African Languages for Automatic Speech Recognition: a Case Study of Wolof","date":"2016-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/character-level-neural-translation-for","slug":"character-level-neural-translation-for","title":"Character-Level Neural Translation for Multilingual Media Monitoring in the SUMMA Project","date":"2016-04-05","arxiv_id":"1604.01221","repositories_listed":1,"syntology":null},{"url":null,"slug":"nonverbaltts-a-public-english-corpus-of-text","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","date":"2025-07-17","arxiv_id":"2507.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisperkit-on-device-real-time-asr-with","title":"WhisperKit: On-device Real-time ASR with Billion-Scale Transformers","date":"2025-07-14","arxiv_id":"2507.10860","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-target-speaker-based-overlap","title":"Lightweight Target-Speaker-Based Overlap Transcription for Practical Streaming ASR","date":"2025-06-25","arxiv_id":"2506.20288","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-generated-song-detection-via-lyrics","title":"AI-Generated Song Detection via Lyrics Transcripts","date":"2025-06-23","arxiv_id":"2506.18488","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-spoken-grammatical-error","title":"End-to-End Spoken Grammatical Error Correction","date":"2025-06-23","arxiv_id":"2506.18532","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-transcription-bottleneck-fine","title":"Breaking the Transcription Bottleneck: Fine-tuning ASR Models for Extremely Low-Resource Fieldwork Languages","date":"2025-06-20","arxiv_id":"2506.17459","repositories_listed":0,"syntology":null},{"url":null,"slug":"lm-spt-lm-aligned-semantic-distillation-for","title":"LM-SPT: LM-Aligned Semantic Distillation for Speech Tokenization","date":"2025-06-20","arxiv_id":"2506.16738","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-biases-in","title":"Automatic Speech Recognition Biases in Newcastle English: an Error Analysis","date":"2025-06-19","arxiv_id":"2506.16558","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-practical-aspects-of-end-to-end","title":"Improving Practical Aspects of End-to-End Multi-Talker Speech Recognition for Online and Offline Scenarios","date":"2025-06-17","arxiv_id":"2506.14204","repositories_listed":0,"syntology":null},{"url":"/paper/unifying-streaming-and-non-streaming","slug":"unifying-streaming-and-non-streaming","title":"Unifying Streaming and Non-streaming Zipformer-based ASR","date":"2025-06-17","arxiv_id":"2506.14434","repositories_listed":0,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unifying-streaming-and-non-streaming#ran","syntology_url":"https://syntology.ai/paper/2506.14434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14434"}},"official":null}},{"url":null,"slug":"bi-directional-context-enhanced-speech-large","title":"Bi-directional Context-Enhanced Speech Large Language Models for Multilingual Conversational ASR","date":"2025-06-16","arxiv_id":"2506.13396","repositories_listed":0,"syntology":null},{"url":null,"slug":"but-system-for-the-mlc-slm-challenge","title":"BUT System for the MLC-SLM Challenge","date":"2025-06-16","arxiv_id":"2506.13414","repositories_listed":0,"syntology":null},{"url":null,"slug":"seewo-s-submission-to-mlc-slm-lessons-learned","title":"Seewo's Submission to MLC-SLM: Lessons learned from Speech Reasoning Language Models","date":"2025-06-16","arxiv_id":"2506.13300","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-automatic-transcription-of-child","title":"Enabling automatic transcription of child-centered audio recordings from real-world environments","date":"2025-06-13","arxiv_id":"2506.11747","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-and-robust-multi-channel-end-to","title":"Lightweight and Robust Multi-Channel End-to-End Speech Recognition with Spherical Harmonic Transform","date":"2025-06-13","arxiv_id":"2506.11630","repositories_listed":0,"syntology":null},{"url":null,"slug":"simphon-speech-test-a-data-driven-method-for","title":"(SimPhon Speech Test): A Data-Driven Method for In Silico Design and Validation of a Phonetically Balanced Speech Test","date":"2025-06-13","arxiv_id":"2506.11620","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-named-entity-transcription-with","title":"Improving Named Entity Transcription with Contextual LLM-based Revision","date":"2025-06-12","arxiv_id":"2506.10779","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-learnable-feature-extraction-for","title":"Regularizing Learnable Feature Extraction for Automatic Speech Recognition","date":"2025-06-11","arxiv_id":"2506.09804","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-foundation-speech-and-language","title":"Benchmarking Foundation Speech and Language Models for Alzheimer's Disease and Related Dementia Detection from Spontaneous Speech","date":"2025-06-09","arxiv_id":"2506.11119","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcript-prompted-whisper-with-dictionary","title":"Transcript-Prompted Whisper with Dictionary-Enhanced Decoding for Japanese Speech Annotation","date":"2025-06-09","arxiv_id":"2506.07646","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-on-tv-series-with-video","title":"Speech Recognition on TV Series with Video-guided Post-Correction","date":"2025-06-08","arxiv_id":"2506.07323","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-of-african","title":"Automatic Speech Recognition of African American English: Lexical and Contextual Effects","date":"2025-06-07","arxiv_id":"2506.06888","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-modality-gap-softly-discretizing","title":"Bridging the Modality Gap: Softly Discretizing Audio Representation for LLM-based Automatic Speech Recognition","date":"2025-06-06","arxiv_id":"2506.05706","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-prompt-biasing-for-contextualized","title":"Lightweight Prompt Biasing for Contextualized End-to-End ASR Systems","date":"2025-06-06","arxiv_id":"2506.06252","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-domain-adaptation-for-speech","title":"Low-Resource Domain Adaptation for Speech LLMs via Text-Only Fine-Tuning","date":"2025-06-06","arxiv_id":"2506.05671","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-pseudo-labeling-with-multi-asr-fusion","title":"Better Pseudo-labeling with Multi-ASR Fusion and Error Correction by SpeechLLM","date":"2025-06-05","arxiv_id":"2506.11089","repositories_listed":0,"syntology":null},{"url":null,"slug":"customizing-speech-recognition-model-with","title":"Customizing Speech Recognition Model with Large Language Model Feedback","date":"2025-06-05","arxiv_id":"2506.11091","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-large-language-model-enhanced-semi","title":"LESS: Large Language Model Enhanced Semi-Supervised Learning for Speech Foundational Models","date":"2025-06-05","arxiv_id":"2506.04586","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-phoneme-to-grapheme-for-phoneme","title":"LLM-based phoneme-to-grapheme for phoneme-based speech recognition","date":"2025-06-05","arxiv_id":"2506.04711","repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-speaker-count-duration-and-accent","title":"Effects of Speaker Count, Duration, and Accent Diversity on Zero-Shot Accent Robustness in Low-Resource ASR","date":"2025-06-04","arxiv_id":"2506.04364","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-dialectal-dataset-for-german-dialect","title":"A Multi-Dialectal Dataset for German Dialect ASR and Dialect-to-Standard Speech Translation","date":"2025-06-03","arxiv_id":"2506.02894","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-lyrics-transcription-on-music","title":"Enhancing Lyrics Transcription on Music Mixtures with Consistency Loss","date":"2025-06-03","arxiv_id":"2506.02339","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-data-scarcity-in-multi-dialectal","title":"Overcoming Data Scarcity in Multi-Dialectal Arabic ASR via Whisper Fine-Tuning","date":"2025-06-03","arxiv_id":"2506.02627","repositories_listed":0,"syntology":null},{"url":null,"slug":"dncasr-end-to-end-training-for-speaker","title":"DNCASR: End-to-End Training for Speaker-Attributed ASR","date":"2025-06-02","arxiv_id":"2506.01916","repositories_listed":0,"syntology":null},{"url":null,"slug":"hent-srt-hierarchical-efficient-neural","title":"HENT-SRT: Hierarchical Efficient Neural Transducer with Self-Distillation for Joint Speech Recognition and Translation","date":"2025-06-02","arxiv_id":"2506.02157","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-structure-discovery-for-error","title":"Causal Structure Discovery for Error Diagnostics of Children's ASR","date":"2025-05-31","arxiv_id":"2506.00402","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-context-aware-streaming-pretrained","title":"Dynamic Context-Aware Streaming Pretrained Language Model For Inverse Text Normalization","date":"2025-05-30","arxiv_id":"2505.24229","repositories_listed":0,"syntology":null},{"url":null,"slug":"fewer-hallucinations-more-verification-a","title":"Fewer Hallucinations, More Verification: A Three-Stage LLM-Based Framework for ASR Error Correction","date":"2025-05-30","arxiv_id":"2505.24347","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multilingual-speech-models-on-ml","title":"Improving Multilingual Speech Models on ML-SUPERB 2.0: Fine-tuning with Data Augmentation and LID-Aware CTC","date":"2025-05-30","arxiv_id":"2505.24200","repositories_listed":0,"syntology":null},{"url":null,"slug":"msda-combining-pseudo-labeling-and-self","title":"MSDA: Combining Pseudo-labeling and Self-Supervision for Unsupervised Domain Adaptation in ASR","date":"2025-05-30","arxiv_id":"2505.24656","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-automatic-speech-recognition-2","title":"Contextualized Automatic Speech Recognition with Dynamic Vocabulary Prediction and Activation","date":"2025-05-29","arxiv_id":"2505.23077","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-whisper-for-improved-verbatim","title":"Prompting Whisper for Improved Verbatim Transcription and End-to-end Miscue Detection","date":"2025-05-29","arxiv_id":"2505.23627","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-hearing-assessment-an-asr-based","title":"Advancing Hearing Assessment: An ASR-Based Frequency-Specific Speech Test for Diagnosing Presbycusis","date":"2025-05-28","arxiv_id":"2505.22231","repositories_listed":0,"syntology":null},{"url":null,"slug":"ngpu-lm-gpu-accelerated-n-gram-language-model","title":"NGPU-LM: GPU-Accelerated N-Gram Language Model for Context-Biasing in Greedy ASR Decoding","date":"2025-05-28","arxiv_id":"2505.22857","repositories_listed":0,"syntology":null},{"url":null,"slug":"loquacious-set-25000-hours-of-transcribed-and","title":"Loquacious Set: 25,000 Hours of Transcribed and Diverse English Speech Recognition Data for Research and Commercial Use","date":"2025-05-27","arxiv_id":"2505.21578","repositories_listed":0,"syntology":null},{"url":null,"slug":"psrb-a-comprehensive-benchmark-for-evaluating","title":"PSRB: A Comprehensive Benchmark for Evaluating Persian ASR Systems","date":"2025-05-27","arxiv_id":"2505.21230","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-pretraining-robust-asr-foundation","title":"Towards Pretraining Robust ASR Foundation Model with Acoustic-Aware Data Augmentation","date":"2025-05-27","arxiv_id":"2505.20606","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-manual-transcripts-the-potential-of","title":"Beyond Manual Transcripts: The Potential of Automated Speech Recognition Errors in Improving Alzheimer's Disease Detection","date":"2025-05-26","arxiv_id":"2505.19448","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-learning-for-children-s-asr","title":"Continuous Learning for Children's ASR: Overcoming Catastrophic Forgetting with Elastic Weight Consolidation and Synaptic Intelligence","date":"2025-05-26","arxiv_id":"2505.20216","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-language-learning-for-endangered","title":"In-context Language Learning for Endangered Languages in Speech Recognition","date":"2025-05-26","arxiv_id":"2505.20445","repositories_listed":0,"syntology":null},{"url":null,"slug":"kit-s-low-resource-speech-translation-systems","title":"KIT's Low-resource Speech Translation Systems for IWSLT2025: System Enhancement with Synthetic Data and Model Regularization","date":"2025-05-26","arxiv_id":"2505.19679","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-lora-experts-for-low-resourced","title":"Mixture of LoRA Experts for Low-Resourced Multi-Accent Automatic Speech Recognition","date":"2025-05-26","arxiv_id":"2505.20006","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-fine-tuning-of-speech-recognition","title":"Robust fine-tuning of speech recognition models via model merging: application to disordered speech","date":"2025-05-26","arxiv_id":"2505.20477","repositories_listed":0,"syntology":null},{"url":null,"slug":"vietasr-achieving-industry-level-vietnamese","title":"VietASR: Achieving Industry-level Vietnamese ASR with 50-hour labeled data and Large-Scale Speech Pretraining","date":"2025-05-23","arxiv_id":"2505.21527","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-training-framework-for-light","title":"An Effective Training Framework for Light-Weight Automatic Speech Recognition Models","date":"2025-05-22","arxiv_id":"2505.16991","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-based-asr-error","title":"Large Language Models based ASR Error Correction for Child Conversations","date":"2025-05-22","arxiv_id":"2505.16212","repositories_listed":0,"syntology":null},{"url":null,"slug":"soccerchat-integrating-multimodal-data-for","title":"SoccerChat: Integrating Multimodal Data for Enhanced Soccer Game Understanding","date":"2025-05-22","arxiv_id":"2505.16630","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-weak-labels-to-strong-results-utilizing","title":"From Weak Labels to Strong Results: Utilizing 5,000 Hours of Noisy Classroom Transcripts with Minimal Accurate Data","date":"2025-05-20","arxiv_id":"2505.17088","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-learning-boosts-speech-recognition","title":"In-Context Learning Boosts Speech Recognition via Human-like Adaptation to Speakers and Language Varieties","date":"2025-05-20","arxiv_id":"2505.14887","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10975","title":"Survey of End-to-End Multi-Speaker Automatic Speech Recognition for Monaural Audio","date":"2025-05-16","arxiv_id":"2505.10975","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11352","title":"LegoSLM: Connecting LLM with Speech Encoder using CTC Posteriors","date":"2025-05-16","arxiv_id":"2505.11352","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-fairbench-measuring-and-benchmarking","title":"ASR-FAIRBENCH: Measuring and Benchmarking Equity Across Speech Recognition Systems","date":"2025-05-16","arxiv_id":"2505.11572","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-african-low","title":"Automatic Speech Recognition for African Low-Resource Languages: Challenges and Future Directions","date":"2025-05-16","arxiv_id":"2505.11690","repositories_listed":0,"syntology":null},{"url":null,"slug":"lipdiffuser-lip-to-speech-generation-with","title":"LipDiffuser: Lip-to-Speech Generation with Conditional Diffusion Models","date":"2025-05-16","arxiv_id":"2505.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"remote-rowhammer-attack-using-adversarial","title":"Remote Rowhammer Attack using Adversarial Observations on Federated Learning Clients","date":"2025-05-09","arxiv_id":"2505.06335","repositories_listed":0,"syntology":null},{"url":null,"slug":"teochew-wild-the-first-in-the-wild-teochew","title":"Teochew-Wild: The First In-the-wild Teochew Dataset with Orthographic Annotations","date":"2025-05-08","arxiv_id":"2505.05056","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-of-automatic-speech-recognition-in","title":"Fairness of Automatic Speech Recognition in Cleft Lip and Palate Speech","date":"2025-05-06","arxiv_id":"2505.03697","repositories_listed":0,"syntology":null},{"url":null,"slug":"sepalm-audio-language-models-are-error","title":"SepALM: Audio Language Models Are Error Correctors for Robust Speech Separation","date":"2025-05-06","arxiv_id":"2505.03273","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-based-deep-residual","title":"Transfer Learning-Based Deep Residual Learning for Speech Recognition in Clean and Noisy Environments","date":"2025-05-02","arxiv_id":"2505.01632","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-enhanced-few-shot-prompting-for","title":"Retrieval-Enhanced Few-Shot Prompting for Speech Event Extraction","date":"2025-04-30","arxiv_id":"2504.21372","repositories_listed":0,"syntology":null},{"url":null,"slug":"chinese-lips-a-chinese-audio-visual-speech","title":"Chinese-LiPS: A Chinese audio-visual speech recognition dataset with Lip-reading and Presentation Slides","date":"2025-04-21","arxiv_id":"2504.15066","repositories_listed":0,"syntology":null},{"url":null,"slug":"stablequant-layer-adaptive-post-training","title":"StableQuant: Layer Adaptive Post-Training Quantization for Speech Foundation Models","date":"2025-04-21","arxiv_id":"2504.14915","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-to-articulatory-inversion-of-speech","title":"Acoustic to Articulatory Inversion of Speech; Data Driven Approaches, Challenges, Applications, and Future Scope","date":"2025-04-17","arxiv_id":"2504.13308","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-arabic-speech-recognition-through","title":"Advancing Arabic Speech Recognition Through Large-Scale Weakly Supervised Learning","date":"2025-04-16","arxiv_id":"2504.12254","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-audio-processing-with-large-language","title":"Spatial Audio Processing with Large Language Model on Wearable Devices","date":"2025-04-11","arxiv_id":"2504.08907","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-aware-speech-recognition-for-noisy","title":"Visual-Aware Speech Recognition for Noisy Scenarios","date":"2025-04-09","arxiv_id":"2504.07229","repositories_listed":0,"syntology":null},{"url":null,"slug":"linto-audio-and-textual-datasets-to-train-and","title":"LinTO Audio and Textual Datasets to Train and Evaluate Automatic Speech Recognition in Tunisian Arabic Dialect","date":"2025-04-03","arxiv_id":"2504.02604","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-correction-for-full-text-speech","title":"Chain of Correction for Full-text Speech Recognition with Large Language Models","date":"2025-04-02","arxiv_id":"2504.01519","repositories_listed":0,"syntology":null}],"record_sha256":"d7865144d09e6ec3be08737bef18f35cb9e30f6593917009300e354c3c0318a2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}