{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/15","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":65,"rows_per_page":100,"rows":[1401,1500],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/14","next":"/task/speech-recognition/papers/16","papers":[{"url":null,"slug":"enabling-automatic-transcription-of-child","title":"Enabling automatic transcription of child-centered audio recordings from real-world environments","date":"2025-06-13","arxiv_id":"2506.11747","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-and-robust-multi-channel-end-to","title":"Lightweight and Robust Multi-Channel End-to-End Speech Recognition with Spherical Harmonic Transform","date":"2025-06-13","arxiv_id":"2506.11630","repositories_listed":0,"syntology":null},{"url":null,"slug":"simphon-speech-test-a-data-driven-method-for","title":"(SimPhon Speech Test): A Data-Driven Method for In Silico Design and Validation of a Phonetically Balanced Speech Test","date":"2025-06-13","arxiv_id":"2506.11620","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairasr-fair-audio-contrastive-learning-for","title":"FairASR: Fair Audio Contrastive Learning for Automatic Speech Recognition","date":"2025-06-12","arxiv_id":"2506.10747","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-named-entity-transcription-with","title":"Improving Named Entity Transcription with Contextual LLM-based Revision","date":"2025-06-12","arxiv_id":"2506.10779","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-asr-and-speaker-role-tagging-with","title":"Joint ASR and Speaker Role Tagging with Serialized Output Training","date":"2025-06-12","arxiv_id":"2506.10349","repositories_listed":0,"syntology":null},{"url":null,"slug":"owsm-biasing-contextualizing-open-whisper","title":"OWSM-Biasing: Contextualizing Open Whisper-Style Speech Models for Automatic Speech Recognition with Dynamic Vocabulary","date":"2025-06-11","arxiv_id":"2506.09448","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-learnable-feature-extraction-for","title":"Regularizing Learnable Feature Extraction for Automatic Speech Recognition","date":"2025-06-11","arxiv_id":"2506.09804","repositories_listed":0,"syntology":null},{"url":null,"slug":"simclass-a-classroom-speech-dataset-generated","title":"SimClass: A Classroom Speech Dataset Generated via Game Engine Simulation For Automatic Speech Recognition Research","date":"2025-06-10","arxiv_id":"2506.09206","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-foundation-speech-and-language","title":"Benchmarking Foundation Speech and Language Models for Alzheimer's Disease and Related Dementia Detection from Spontaneous Speech","date":"2025-06-09","arxiv_id":"2506.11119","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-distinguishable-ctc-learning-speaker","title":"Speaker-Distinguishable CTC: Learning Speaker Distinction Using CTC for Multi-Talker Speech Recognition","date":"2025-06-09","arxiv_id":"2506.07515","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcript-prompted-whisper-with-dictionary","title":"Transcript-Prompted Whisper with Dictionary-Enhanced Decoding for Japanese Speech Annotation","date":"2025-06-09","arxiv_id":"2506.07646","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncovering-the-functional-roles-of","title":"Uncovering the Functional Roles of Nonlinearity in Memory","date":"2025-06-09","arxiv_id":"2506.07919","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-semi-supervised-pipeline-for","title":"Unified Semi-Supervised Pipeline for Automatic Speech Recognition","date":"2025-06-09","arxiv_id":"2506.07659","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-on-tv-series-with-video","title":"Speech Recognition on TV Series with Video-guided Post-Correction","date":"2025-06-08","arxiv_id":"2506.07323","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-of-african","title":"Automatic Speech Recognition of African American English: Lexical and Contextual Effects","date":"2025-06-07","arxiv_id":"2506.06888","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-classification-towards-speech-emotion","title":"Beyond Classification: Towards Speech Emotion Reasoning with Multitask AudioLLMs","date":"2025-06-07","arxiv_id":"2506.06820","repositories_listed":0,"syntology":null},{"url":null,"slug":"as-asr-a-lightweight-framework-for-aphasia","title":"AS-ASR: A Lightweight Framework for Aphasia-Specific Automatic Speech Recognition","date":"2025-06-06","arxiv_id":"2506.06566","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-modality-gap-softly-discretizing","title":"Bridging the Modality Gap: Softly Discretizing Audio Representation for LLM-based Automatic Speech Recognition","date":"2025-06-06","arxiv_id":"2506.05706","repositories_listed":0,"syntology":null},{"url":null,"slug":"diarization-aware-multi-speaker-automatic","title":"Diarization-Aware Multi-Speaker Automatic Speech Recognition via Large Language Models","date":"2025-06-06","arxiv_id":"2506.05796","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-prompt-biasing-for-contextualized","title":"Lightweight Prompt Biasing for Contextualized End-to-End ASR Systems","date":"2025-06-06","arxiv_id":"2506.06252","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-domain-adaptation-for-speech","title":"Low-Resource Domain Adaptation for Speech LLMs via Text-Only Fine-Tuning","date":"2025-06-06","arxiv_id":"2506.05671","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-pseudo-labeling-with-multi-asr-fusion","title":"Better Pseudo-labeling with Multi-ASR Fusion and Error Correction by SpeechLLM","date":"2025-06-05","arxiv_id":"2506.11089","repositories_listed":0,"syntology":null},{"url":null,"slug":"customizing-speech-recognition-model-with","title":"Customizing Speech Recognition Model with Large Language Model Feedback","date":"2025-06-05","arxiv_id":"2506.11091","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-large-language-model-enhanced-semi","title":"LESS: Large Language Model Enhanced Semi-Supervised Learning for Speech Foundational Models","date":"2025-06-05","arxiv_id":"2506.04586","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-phoneme-to-grapheme-for-phoneme","title":"LLM-based phoneme-to-grapheme for phoneme-based speech recognition","date":"2025-06-05","arxiv_id":"2506.04711","repositories_listed":0,"syntology":null},{"url":null,"slug":"vicocktail-automated-multi-modal-data","title":"ViCocktail: Automated Multi-Modal Data Collection for Vietnamese Audio-Visual Speech Recognition","date":"2025-06-05","arxiv_id":"2506.04635","repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-speaker-count-duration-and-accent","title":"Effects of Speaker Count, Duration, and Accent Diversity on Zero-Shot Accent Robustness in Low-Resource ASR","date":"2025-06-04","arxiv_id":"2506.04364","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-child-speech-recognition-and","title":"Improving Child Speech Recognition and Reading Mistake Detection by Using Prompts","date":"2025-06-04","arxiv_id":"2506.11079","repositories_listed":0,"syntology":null},{"url":null,"slug":"mfla-monotonic-finite-look-ahead-attention","title":"MFLA: Monotonic Finite Look-ahead Attention for Streaming Speech Recognition","date":"2025-06-04","arxiv_id":"2506.03722","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-dialectal-dataset-for-german-dialect","title":"A Multi-Dialectal Dataset for German Dialect ASR and Dialect-to-Standard Speech Translation","date":"2025-06-03","arxiv_id":"2506.02894","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-lyrics-transcription-on-music","title":"Enhancing Lyrics Transcription on Music Mixtures with Consistency Loss","date":"2025-06-03","arxiv_id":"2506.02339","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-data-scarcity-in-multi-dialectal","title":"Overcoming Data Scarcity in Multi-Dialectal Arabic ASR via Whisper Fine-Tuning","date":"2025-06-03","arxiv_id":"2506.02627","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-importance-of-blank-for-ctc","title":"Analyzing the Importance of Blank for CTC-Based Knowledge Distillation","date":"2025-06-02","arxiv_id":"2506.01503","repositories_listed":0,"syntology":null},{"url":null,"slug":"cocktail-party-audio-visual-speech","title":"Cocktail-Party Audio-Visual Speech Recognition","date":"2025-06-02","arxiv_id":"2506.02178","repositories_listed":0,"syntology":null},{"url":null,"slug":"dncasr-end-to-end-training-for-speaker","title":"DNCASR: End-to-End Training for Speaker-Attributed ASR","date":"2025-06-02","arxiv_id":"2506.01916","repositories_listed":0,"syntology":null},{"url":null,"slug":"hent-srt-hierarchical-efficient-neural","title":"HENT-SRT: Hierarchical Efficient Neural Transducer with Self-Distillation for Joint Speech Recognition and Translation","date":"2025-06-02","arxiv_id":"2506.02157","repositories_listed":0,"syntology":null},{"url":null,"slug":"riemannian-time-warping-multiple-sequence","title":"Riemannian Time Warping: Multiple Sequence Alignment in Curved Spaces","date":"2025-06-02","arxiv_id":"2506.01635","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-speech-quality-assessment","title":"Self-Supervised Speech Quality Assessment (S3QA): Leveraging Speech Foundation Models for a Scalable Speech Quality Metric","date":"2025-06-02","arxiv_id":"2506.01655","repositories_listed":0,"syntology":null},{"url":null,"slug":"taltech-systems-for-the-interspeech-2025-ml","title":"TalTech Systems for the Interspeech 2025 ML-SUPERB 2.0 Challenge","date":"2025-06-02","arxiv_id":"2506.01458","repositories_listed":0,"syntology":null},{"url":null,"slug":"wctc-biasing-retraining-free-contextual","title":"WCTC-Biasing: Retraining-free Contextual Biasing ASR with Wildcard CTC-based Keyword Spotting and Inter-layer Biasing","date":"2025-06-02","arxiv_id":"2506.01263","repositories_listed":0,"syntology":null},{"url":null,"slug":"whale-large-scale-multilingual-asr-model-with","title":"Whale: Large-Scale multilingual ASR model with w2v-BERT and E-Branchformer with large speech data","date":"2025-06-02","arxiv_id":"2506.01439","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-instruction-understanding","title":"Enhancing Speech Instruction Understanding and Disambiguation in Robotics via Speech Prosody","date":"2025-06-01","arxiv_id":"2506.02057","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-structure-discovery-for-error","title":"Causal Structure Discovery for Error Diagnostics of Children's ASR","date":"2025-05-31","arxiv_id":"2506.00402","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-thought-training-for-open-e2e-spoken","title":"Chain-of-Thought Training for Open E2E Spoken Dialogue Systems","date":"2025-05-31","arxiv_id":"2506.00722","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynac-dynamic-vocabulary-based-non","title":"DYNAC: Dynamic Vocabulary based Non-Autoregressive Contextualization for Speech Recognition","date":"2025-05-31","arxiv_id":"2506.00422","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-audiogram-leveraging-existing-scores-for","title":"No Audiogram: Leveraging Existing Scores for Personalized Speech Intelligibility Prediction","date":"2025-05-31","arxiv_id":"2506.02039","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-context-aware-streaming-pretrained","title":"Dynamic Context-Aware Streaming Pretrained Language Model For Inverse Text Normalization","date":"2025-05-30","arxiv_id":"2505.24229","repositories_listed":0,"syntology":null},{"url":null,"slug":"fewer-hallucinations-more-verification-a","title":"Fewer Hallucinations, More Verification: A Three-Stage LLM-Based Framework for ASR Error Correction","date":"2025-05-30","arxiv_id":"2505.24347","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multilingual-speech-models-on-ml","title":"Improving Multilingual Speech Models on ML-SUPERB 2.0: Fine-tuning with Data Augmentation and LID-Aware CTC","date":"2025-05-30","arxiv_id":"2505.24200","repositories_listed":0,"syntology":null},{"url":null,"slug":"mopsa-mixture-of-prompt-experts-based-speaker","title":"MOPSA: Mixture of Prompt-Experts Based Speaker Adaptation for Elderly Speech Recognition","date":"2025-05-30","arxiv_id":"2505.24224","repositories_listed":0,"syntology":null},{"url":null,"slug":"msda-combining-pseudo-labeling-and-self","title":"MSDA: Combining Pseudo-labeling and Self-Supervision for Unsupervised Domain Adaptation in ASR","date":"2025-05-30","arxiv_id":"2505.24656","repositories_listed":0,"syntology":null},{"url":null,"slug":"running-conventional-automatic-speech","title":"Running Conventional Automatic Speech Recognition on Memristor Hardware: A Simulated Approach","date":"2025-05-30","arxiv_id":"2505.24721","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-automatic-speech-recognition-2","title":"Contextualized Automatic Speech Recognition with Dynamic Vocabulary Prediction and Activation","date":"2025-05-29","arxiv_id":"2505.23077","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-whisper-for-improved-verbatim","title":"Prompting Whisper for Improved Verbatim Transcription and End-to-end Miscue Detection","date":"2025-05-29","arxiv_id":"2505.23627","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-hearing-assessment-an-asr-based","title":"Advancing Hearing Assessment: An ASR-Based Frequency-Specific Speech Test for Diagnosing Presbycusis","date":"2025-05-28","arxiv_id":"2505.22231","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-llms-in-speech-is-often-flawed","title":"Evaluation of LLMs in Speech is Often Flawed: Test Set Contamination in Large Language Models for Speech Recognition","date":"2025-05-28","arxiv_id":"2505.22251","repositories_listed":0,"syntology":null},{"url":null,"slug":"ngpu-lm-gpu-accelerated-n-gram-language-model","title":"NGPU-LM: GPU-Accelerated N-Gram Language Model for Context-Biasing in Greedy ASR Decoding","date":"2025-05-28","arxiv_id":"2505.22857","repositories_listed":0,"syntology":null},{"url":null,"slug":"cnvsrc-2024-the-second-chinese-continuous","title":"CNVSRC 2024: The Second Chinese Continuous Visual Speech Recognition Challenge","date":"2025-05-27","arxiv_id":"2506.02010","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-language-models-in-visual","title":"Leveraging Large Language Models in Visual Speech Recognition: Model Scaling, Context-Aware Decoding, and Iterative Polishing","date":"2025-05-27","arxiv_id":"2506.02012","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-llm-and-self-supervised-training","title":"Leveraging LLM and Self-Supervised Training Models for Speech Recognition in Chinese Dialects: A Comparative Analysis","date":"2025-05-27","arxiv_id":"2505.21138","repositories_listed":0,"syntology":null},{"url":null,"slug":"loquacious-set-25000-hours-of-transcribed-and","title":"Loquacious Set: 25,000 Hours of Transcribed and Diverse English Speech Recognition Data for Research and Commercial Use","date":"2025-05-27","arxiv_id":"2505.21578","repositories_listed":0,"syntology":null},{"url":null,"slug":"psrb-a-comprehensive-benchmark-for-evaluating","title":"PSRB: A Comprehensive Benchmark for Evaluating Persian ASR Systems","date":"2025-05-27","arxiv_id":"2505.21230","repositories_listed":0,"syntology":null},{"url":null,"slug":"topological-deep-learning-for-speech-data","title":"Topological Deep Learning for Speech Data","date":"2025-05-27","arxiv_id":"2505.21173","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-pretraining-robust-asr-foundation","title":"Towards Pretraining Robust ASR Foundation Model with Acoustic-Aware Data Augmentation","date":"2025-05-27","arxiv_id":"2505.20606","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-manual-transcripts-the-potential-of","title":"Beyond Manual Transcripts: The Potential of Automated Speech Recognition Errors in Improving Alzheimer's Disease Detection","date":"2025-05-26","arxiv_id":"2505.19448","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-learning-for-children-s-asr","title":"Continuous Learning for Children's ASR: Overcoming Catastrophic Forgetting with Elastic Weight Consolidation and Synaptic Intelligence","date":"2025-05-26","arxiv_id":"2505.20216","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-language-learning-for-endangered","title":"In-context Language Learning for Endangered Languages in Speech Recognition","date":"2025-05-26","arxiv_id":"2505.20445","repositories_listed":0,"syntology":null},{"url":null,"slug":"kit-s-low-resource-speech-translation-systems","title":"KIT's Low-resource Speech Translation Systems for IWSLT2025: System Enhancement with Synthetic Data and Model Regularization","date":"2025-05-26","arxiv_id":"2505.19679","repositories_listed":0,"syntology":null},{"url":null,"slug":"languages-in-multilingual-speech-foundation","title":"Languages in Multilingual Speech Foundation Models Align Both Phonetically and Semantically","date":"2025-05-26","arxiv_id":"2505.19606","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-lora-experts-for-low-resourced","title":"Mixture of LoRA Experts for Low-Resourced Multi-Accent Automatic Speech Recognition","date":"2025-05-26","arxiv_id":"2505.20006","repositories_listed":0,"syntology":null},{"url":null,"slug":"novel-loss-enhanced-universal-adversarial","title":"Novel Loss-Enhanced Universal Adversarial Patches for Sustainable Speaker Privacy","date":"2025-05-26","arxiv_id":"2505.19951","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-fine-tuning-of-speech-recognition","title":"Robust fine-tuning of speech recognition models via model merging: application to disordered speech","date":"2025-05-26","arxiv_id":"2505.20477","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-naijavoices-dataset-cultivating-large","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","date":"2025-05-26","arxiv_id":"2505.20564","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisperd-dementia-speech-recognition-and","title":"WhisperD: Dementia Speech Recognition and Filler Word Detection with Whisper","date":"2025-05-25","arxiv_id":"2505.21551","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-functional-machine-translation","title":"Building a Functional Machine Translation Corpus for Kpelle","date":"2025-05-24","arxiv_id":"2505.18905","repositories_listed":0,"syntology":null},{"url":null,"slug":"standup4ai-a-new-multilingual-dataset-for","title":"StandUp4AI: A New Multilingual Dataset for Humor Detection in Stand-up Comedy Videos","date":"2025-05-24","arxiv_id":"2505.18903","repositories_listed":0,"syntology":null},{"url":null,"slug":"swedish-whispers-leveraging-a-massive-speech","title":"Swedish Whispers; Leveraging a Massive Speech Corpus for Swedish Speech Recognition","date":"2025-05-23","arxiv_id":"2505.17538","repositories_listed":0,"syntology":null},{"url":null,"slug":"vietasr-achieving-industry-level-vietnamese","title":"VietASR: Achieving Industry-level Vietnamese ASR with 50-hour labeled data and Large-Scale Speech Pretraining","date":"2025-05-23","arxiv_id":"2505.21527","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-training-framework-for-light","title":"An Effective Training Framework for Light-Weight Automatic Speech Recognition Models","date":"2025-05-22","arxiv_id":"2505.16991","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-based-asr-error","title":"Large Language Models based ASR Error Correction for Child Conversations","date":"2025-05-22","arxiv_id":"2505.16212","repositories_listed":0,"syntology":null},{"url":null,"slug":"soccerchat-integrating-multimodal-data-for","title":"SoccerChat: Integrating Multimodal Data for Enhanced Soccer Game Understanding","date":"2025-05-22","arxiv_id":"2505.16630","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-weak-labels-to-strong-results-utilizing","title":"From Weak Labels to Strong Results: Utilizing 5,000 Hours of Noisy Classroom Transcripts with Minimal Accurate Data","date":"2025-05-20","arxiv_id":"2505.17088","repositories_listed":0,"syntology":null},{"url":null,"slug":"hausanlp-current-status-challenges-and-future","title":"HausaNLP: Current Status, Challenges and Future Directions for Hausa Natural Language Processing","date":"2025-05-20","arxiv_id":"2505.14311","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-frame-rates-on-speech-tokenizer-a","title":"Impact of Frame Rates on Speech Tokenizer: A Case Study on Mandarin and English","date":"2025-05-20","arxiv_id":"2505.17076","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-learning-boosts-speech-recognition","title":"In-Context Learning Boosts Speech Recognition via Human-like Adaptation to Speakers and Language Varieties","date":"2025-05-20","arxiv_id":"2505.14887","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-and-enhancing-llm-based-avsr-a-sparse","title":"Scaling and Enhancing LLM-based AVSR: A Sparse Mixture of Projectors Approach","date":"2025-05-20","arxiv_id":"2505.14336","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multimodal-information-based-speech-2","title":"The Multimodal Information Based Speech Processing (MISP) 2025 Challenge: Audio-Visual Diarization and Recognition","date":"2025-05-20","arxiv_id":"2505.13971","repositories_listed":0,"syntology":null},{"url":null,"slug":"calm-whisper-reduce-whisper-hallucination-on","title":"Calm-Whisper: Reduce Whisper Hallucination On Non-Speech By Calming Crazy Heads Down","date":"2025-05-19","arxiv_id":"2505.12969","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-knowledge-transfer-learning-as","title":"Cross-modal Knowledge Transfer Learning as Graph Matching Based on Optimal Transport for ASR","date":"2025-05-19","arxiv_id":"2505.13079","repositories_listed":0,"syntology":null},{"url":null,"slug":"granary-speech-recognition-and-translation","title":"Granary: Speech Recognition and Translation Dataset in 25 European Languages","date":"2025-05-19","arxiv_id":"2505.13404","repositories_listed":0,"syntology":null},{"url":null,"slug":"kit-s-offline-speech-translation-and","title":"KIT's Offline Speech Translation and Instruction Following Submission for IWSLT 2025","date":"2025-05-19","arxiv_id":"2505.13036","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10975","title":"Survey of End-to-End Multi-Speaker Automatic Speech Recognition for Monaural Audio","date":"2025-05-16","arxiv_id":"2505.10975","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11352","title":"LegoSLM: Connecting LLM with Speech Encoder using CTC Posteriors","date":"2025-05-16","arxiv_id":"2505.11352","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-fairbench-measuring-and-benchmarking","title":"ASR-FAIRBENCH: Measuring and Benchmarking Equity Across Speech Recognition Systems","date":"2025-05-16","arxiv_id":"2505.11572","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-african-low","title":"Automatic Speech Recognition for African Low-Resource Languages: Challenges and Future Directions","date":"2025-05-16","arxiv_id":"2505.11690","repositories_listed":0,"syntology":null},{"url":null,"slug":"lipdiffuser-lip-to-speech-generation-with","title":"LipDiffuser: Lip-to-Speech Generation with Conditional Diffusion Models","date":"2025-05-16","arxiv_id":"2505.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"inclusivity-of-ai-speech-in-healthcare-a","title":"Inclusivity of AI Speech in Healthcare: A Decade Look Back","date":"2025-05-15","arxiv_id":"2505.10596","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantized-approximate-signal-processing-qasp","title":"Quantized Approximate Signal Processing (QASP): Towards Homomorphic Encryption for audio","date":"2025-05-15","arxiv_id":"2505.10500","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-simulation-on-the-dynamics-of-auditory","title":"Full simulation on the dynamics of auditory synaptic fusion: Strong clustering of calcium channel might be the origin of the coherent release in the auditory hair cells","date":"2025-05-12","arxiv_id":"2505.07273","repositories_listed":0,"syntology":null}],"record_sha256":"a41a77df07e7a680bc0330a4c7fe4a043cb8c6f7982ff7d80b4394e260632522","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}