{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/8","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":31,"rows_per_page":100,"rows":[701,800],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/7","next":"/task/automatic-speech-recognition/papers/9","papers":[{"url":null,"slug":"the-impact-of-code-switched-synthetic-data","title":"The Impact of Code-switched Synthetic Data Quality is Task Dependent: Insights from MT and ASR","date":"2025-03-30","arxiv_id":"2503.23576","repositories_listed":0,"syntology":null},{"url":null,"slug":"finaudio-a-benchmark-for-audio-large-language","title":"FinAudio: A Benchmark for Audio Large Language Models in Financial Applications","date":"2025-03-26","arxiv_id":"2503.20990","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-the-transferability-of-audio","title":"Boosting the Transferability of Audio Adversarial Examples with Acoustic Representation Optimization","date":"2025-03-25","arxiv_id":"2503.19591","repositories_listed":0,"syntology":null},{"url":null,"slug":"whispering-in-amharic-fine-tuning-whisper-for","title":"Whispering in Amharic: Fine-tuning Whisper for Low-resource Language","date":"2025-03-24","arxiv_id":"2503.18485","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-voice-is-your-voice-supporting-self","title":"Your voice is your voice: Supporting Self-expression through Speech Generation and LLMs in Augmented and Alternative Communication","date":"2025-03-21","arxiv_id":"2503.17479","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-asr-confidence-scores-for","title":"Evaluating ASR Confidence Scores for Automated Error Detection in User-Assisted Correction Interfaces","date":"2025-03-19","arxiv_id":"2503.15124","repositories_listed":0,"syntology":null},{"url":null,"slug":"everything-can-be-described-in-words-a-simple","title":"Everything Can Be Described in Words: A Simple Unified Multi-Modal Framework with Semantic and Temporal Alignment","date":"2025-03-12","arxiv_id":"2503.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"valsub-subsampling-validation-data-to","title":"ValSub: Subsampling Validation Data to Mitigate Forgetting during ASR Personalization","date":"2025-03-12","arxiv_id":"2503.09906","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exhaustive-evaluation-of-tts-and-vc-based","title":"An Exhaustive Evaluation of TTS- and VC-based Data Augmentation for ASR","date":"2025-03-11","arxiv_id":"2503.08954","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-non-native","title":"Automatic Speech Recognition for Non-Native English: Accuracy and Disfluency Handling","date":"2025-03-10","arxiv_id":"2503.06924","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-english-asr-model-with-regional","title":"Building English ASR model with regional language support","date":"2025-03-10","arxiv_id":"2503.07522","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-voice-to-safety-language-ai-powered","title":"From Voice to Safety: Language AI Powered Pilot-ATC Communication Understanding for Airport Surface Movement Collision Risk Assessment","date":"2025-03-06","arxiv_id":"2503.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"qieemo-speech-is-all-you-need-in-the-emotion","title":"Qieemo: Speech Is All You Need in the Emotion Recognition in Conversations","date":"2025-03-05","arxiv_id":"2503.22687","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-speech-to-speech-translation-a-review","title":"Direct Speech to Speech Translation: A Review","date":"2025-03-03","arxiv_id":"2503.04799","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-whisper-for-inclusive-prosodic","title":"Fine-Tuning Whisper for Inclusive Prosodic Stress Analysis","date":"2025-03-03","arxiv_id":"2503.02907","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-biases-while-embracing","title":"Unveiling Biases while Embracing Sustainability: Assessing the Dual Challenges of Automatic Speech Recognition Systems","date":"2025-03-02","arxiv_id":"2503.00907","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-automatic-speech-recognition-for","title":"Adapting Automatic Speech Recognition for Accented Air Traffic Control Communications","date":"2025-02-27","arxiv_id":"2502.20311","repositories_listed":0,"syntology":null},{"url":null,"slug":"cs-dialogue-a-104-hour-dataset-of-spontaneous","title":"CS-Dialogue: A 104-Hour Dataset of Spontaneous Mandarin-English Code-Switching Dialogues for Speech Recognition","date":"2025-02-26","arxiv_id":"2502.18913","repositories_listed":0,"syntology":null},{"url":null,"slug":"nexus-o-an-omni-perceptive-and-interactive","title":"Nexus: An Omni-Perceptive And -Interactive Model for Language, Audio, And Vision","date":"2025-02-26","arxiv_id":"2503.01879","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-gender-disparities-in-automatic","title":"Exploring Gender Disparities in Automatic Speech Recognition Technology","date":"2025-02-25","arxiv_id":"2502.18434","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-zero-shot-rare-word-recognition","title":"Understanding Zero-shot Rare Word Recognition Improvements Through LLM Integration","date":"2025-02-22","arxiv_id":"2502.16142","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-large-language-models-with","title":"Enhancing Speech Large Language Models with Prompt-Aware Mixture of Audio Encoders","date":"2025-02-21","arxiv_id":"2502.15178","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-esethu-framework-reimagining-sustainable","title":"The Esethu Framework: Reimagining Sustainable Dataset Governance and Curation for Low-Resource Languages","date":"2025-02-21","arxiv_id":"2502.15916","repositories_listed":0,"syntology":null},{"url":null,"slug":"adopting-whisper-for-confidence-estimation","title":"Adopting Whisper for Confidence Estimation","date":"2025-02-19","arxiv_id":"2502.13446","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-automatic-speech-recognition","title":"Benchmarking Automatic Speech Recognition coupled LLM Modules for Medical Diagnostics","date":"2025-02-18","arxiv_id":"2502.13982","repositories_listed":0,"syntology":null},{"url":null,"slug":"gesture-aware-zero-shot-speech-recognition","title":"Gesture-Aware Zero-Shot Speech Recognition for Patients with Language Disorders","date":"2025-02-18","arxiv_id":"2502.13983","repositories_listed":0,"syntology":null},{"url":null,"slug":"lost-in-transcription-found-in-distribution","title":"Lost in Transcription, Found in Distribution Shift: Demystifying Hallucination in Speech Foundation Models","date":"2025-02-18","arxiv_id":"2502.12414","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtlm-an-innovative-language-model-training","title":"MTLM: Incorporating Bidirectional Text Information to Enhance Language Model Training in Speech Recognition Systems","date":"2025-02-14","arxiv_id":"2502.10058","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-analysis-of-asr-errors-for-children","title":"Causal Analysis of ASR Errors for Children: Quantifying the Impact of Physiological, Cognitive, and Extrinsic Factors","date":"2025-02-12","arxiv_id":"2502.08587","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-standard-and-dialectal-frisian-asr","title":"Evaluating Standard and Dialectal Frisian ASR: Multilingual Fine-tuning and Language Identification for Improved Low-resource Performance","date":"2025-02-07","arxiv_id":"2502.04883","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrispeech-dialog-a-benchmark-dataset-for","title":"Afrispeech-Dialog: A Benchmark Dataset for Spontaneous English Conversations in Healthcare and Beyond","date":"2025-02-06","arxiv_id":"2502.03945","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-differentiable-alignment-framework-for","title":"A Differentiable Alignment Framework for Sequence-to-Sequence Modeling via Optimal Transport","date":"2025-02-03","arxiv_id":"2502.01588","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-dro-robust-optimization-for-reducing","title":"CTC-DRO: Robust Optimization for Reducing Language Disparities in Speech Recognition","date":"2025-02-03","arxiv_id":"2502.01777","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-mispronunciation-pattern","title":"Data-Driven Mispronunciation Pattern Discovery for Robust Speech Recognition","date":"2025-02-01","arxiv_id":"2502.00583","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-end-to-end-is-overkill-rethinking","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","date":"2025-02-01","arxiv_id":"2502.00377","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-bias-in-self-supervised-learning-for","title":"Language Bias in Self-Supervised Learning For Automatic Speech Recognition","date":"2025-01-31","arxiv_id":"2501.19321","repositories_listed":0,"syntology":null},{"url":null,"slug":"selma-a-speech-enabled-language-model-for","title":"SELMA: A Speech-Enabled Language Model for Virtual Assistant Interactions","date":"2025-01-31","arxiv_id":"2501.19377","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-embedding-clustering-for","title":"Cross-lingual Embedding Clustering for Hierarchical Softmax in Low-Resource Multilingual Speech Recognition","date":"2025-01-29","arxiv_id":"2501.17615","repositories_listed":0,"syntology":null},{"url":null,"slug":"seal-speech-embedding-alignment-learning-for","title":"SEAL: Speech Embedding Alignment Learning for Speech Large Language Model with Retrieval-Augmented Generation","date":"2025-01-26","arxiv_id":"2502.02603","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multicultural-medical-assistant-can-llms","title":"The Multicultural Medical Assistant: Can LLMs Improve Medical ASR Errors Across Borders?","date":"2025-01-25","arxiv_id":"2501.15310","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-compact-phrasal-rewrites-with","title":"Predicting Compact Phrasal Rewrites with Large Language Models for ASR Post Editing","date":"2025-01-23","arxiv_id":"2501.13831","repositories_listed":0,"syntology":null},{"url":"/paper/let-ssms-be-convnets-state-space-modeling","slug":"let-ssms-be-convnets-state-space-modeling","title":"Let SSMs be ConvNets: State-space Modeling with Optimal Tensor Contractions","date":"2025-01-22","arxiv_id":"2501.13230","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-whisper-asr-hallucinations","title":"Investigation of Whisper ASR Hallucinations Induced by Non-Speech Audio","date":"2025-01-20","arxiv_id":"2501.11378","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmark-of-french-asr-systems-based-on","title":"A Benchmark of French ASR Systems Based on Error Severity","date":"2025-01-18","arxiv_id":"2501.10879","repositories_listed":0,"syntology":null},{"url":null,"slug":"gec-rag-improving-generative-error-correction","title":"GEC-RAG: Improving Generative Error Correction via Retrieval-Augmented Generation for Automatic Speech Recognition Systems","date":"2025-01-18","arxiv_id":"2501.10734","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-rhythm-and-voice-conversion-of","title":"Unsupervised Rhythm and Voice Conversion of Dysarthric to Healthy Speech for ASR","date":"2025-01-17","arxiv_id":"2501.10256","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-whisper-for-regional-dialects","title":"Adapting Whisper for Regional Dialects: Enhancing Public Services for Vulnerable Populations in the United Kingdom","date":"2025-01-15","arxiv_id":"2501.08502","repositories_listed":0,"syntology":null},{"url":null,"slug":"persoda-personalized-data-augmentation","title":"persoDA: Personalized Data Augmentation for Personalized ASR","date":"2025-01-15","arxiv_id":"2501.09113","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-for-automatically","title":"Speech Recognition for Automatically Assessing Afrikaans and isiXhosa Preschool Oral Narratives","date":"2025-01-11","arxiv_id":"2501.06478","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-rotary-position-embeddings-for","title":"Benchmarking Rotary Position Embeddings for Automatic Speech Recognition","date":"2025-01-10","arxiv_id":"2501.06051","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-asr-error-handling-with-llms","title":"Contextual ASR Error Handling with LLMs Augmentation for Goal-Oriented Conversational AI","date":"2025-01-10","arxiv_id":"2501.06129","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-2-tf-robust-all-neural-text","title":"Universal-2-TF: Robust All-Neural Text Formatting for ASR","date":"2025-01-10","arxiv_id":"2501.05948","repositories_listed":0,"syntology":null},{"url":"/paper/samba-asr-state-of-the-art-speech-recognition","slug":"samba-asr-state-of-the-art-speech-recognition","title":"Samba-ASR: State-Of-The-Art Speech Recognition Leveraging Structured State-Space Models","date":"2025-01-06","arxiv_id":"2501.02832","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-transducer-based-spoken-language","title":"Improving Transducer-Based Spoken Language Understanding with Self-Conditioned CTC and Knowledge Transfer","date":"2025-01-03","arxiv_id":"2501.01936","repositories_listed":0,"syntology":null},{"url":null,"slug":"livecc-learning-video-llm-with-streaming","title":"LiveCC: Learning Video LLM with Streaming Speech Transcription at Scale","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-resource-speech-translation-and","title":"Zero-resource Speech Translation and Recognition with LLMs","date":"2024-12-24","arxiv_id":"2412.18566","repositories_listed":0,"syntology":null},{"url":null,"slug":"ume-upcycling-mixture-of-experts-for-scalable","title":"UME: Upcycling Mixture-of-Experts for Scalable and Efficient Automatic Speech Recognition","date":"2024-12-23","arxiv_id":"2412.17507","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-whisper-for-code-switching-through","title":"Adapting Whisper for Code-Switching through Encoding Refining and Language-Aware Decoding","date":"2024-12-21","arxiv_id":"2412.16507","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multilingual-asr-for-unseen","title":"Enhancing Multilingual ASR for Unseen Languages via Language Embedding Modeling","date":"2024-12-21","arxiv_id":"2412.16474","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-retrieval-augmented-generation-without","title":"Speech Retrieval-Augmented Generation without Automatic Speech Recognition","date":"2024-12-21","arxiv_id":"2412.16500","repositories_listed":0,"syntology":null},{"url":null,"slug":"transducer-llama-integrating-llms-into","title":"Transducer-Llama: Integrating LLMs into Streamable Transducer-based Speech Recognition","date":"2024-12-21","arxiv_id":"2412.16464","repositories_listed":0,"syntology":null},{"url":null,"slug":"touchasp-elastic-automatic-speech-perception","title":"TouchASP: Elastic Automatic Speech Perception that Everyone Can Touch","date":"2024-12-20","arxiv_id":"2412.15622","repositories_listed":0,"syntology":null},{"url":null,"slug":"lama-ut-language-agnostic-multilingual-asr","title":"LAMA-UT: Language Agnostic Multilingual ASR through Orthography Unification and Language-Specific Transliteration","date":"2024-12-19","arxiv_id":"2412.15299","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribing-and-translating-fast-and-slow","title":"Transcribing and Translating, Fast and Slow: Joint Speech Translation and Recognition","date":"2024-12-19","arxiv_id":"2412.15415","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-improve-challenge-2025-tasks-and","title":"Speak & Improve Challenge 2025: Tasks and Baseline Systems","date":"2024-12-16","arxiv_id":"2412.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-improve-corpus-2025-an-l2-english","title":"Speak & Improve Corpus 2025: an L2 English Speech Corpus for Language Assessment and Feedback","date":"2024-12-16","arxiv_id":"2412.11986","repositories_listed":0,"syntology":null},{"url":null,"slug":"greek2mathtex-a-greek-speech-to-text","title":"Greek2MathTex: A Greek Speech-to-Text Framework for LaTeX Equations Generation","date":"2024-12-11","arxiv_id":"2412.12167","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-text-adaptation-for-llm-based-asr","title":"Effective Text Adaptation for LLM-based ASR through Soft Prompt Fine-Tuning","date":"2024-12-09","arxiv_id":"2412.06967","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-transfer-learning-from-swahili","title":"Harnessing Transfer Learning from Swahili: Advancing Solutions for Comorian Dialects","date":"2024-12-09","arxiv_id":"2412.12143","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-prompt-learning-and-pause-encoding","title":"Leveraging Prompt Learning and Pause Encoding for Alzheimer's Disease Detection","date":"2024-12-09","arxiv_id":"2412.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-all-errors-are-equal-investigation-of","title":"Not All Errors Are Equal: Investigation of Speech Recognition Errors in Alzheimer's Disease Detection","date":"2024-12-09","arxiv_id":"2412.06332","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-audio-query-handling-system","title":"Comprehensive Audio Query Handling System with Integrated Expert Models and Contextual Understanding","date":"2024-12-05","arxiv_id":"2412.03980","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-ec-benchmark-evaluating-large-language","title":"ASR-EC Benchmark: Evaluating Large Language Models on Chinese ASR Error Correction","date":"2024-12-04","arxiv_id":"2412.03075","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-llm-based-asr-and","title":"A Comparative Study of LLM-based ASR and Whisper in Low Resource and Code Switching Scenario","date":"2024-12-01","arxiv_id":"2412.00721","repositories_listed":0,"syntology":null},{"url":null,"slug":"late-fusion-ensembles-for-speech-recognition","title":"Late fusion ensembles for speech recognition on diverse input audio representations","date":"2024-12-01","arxiv_id":"2412.01861","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-the-deaf-and-hard-of-hearing","title":"Empowering the Deaf and Hard of Hearing Community: Enhancing Video Captions Using Large Language Models","date":"2024-11-30","arxiv_id":"2412.00342","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-pre-trained-models-for-spoken","title":"Aligning Pre-trained Models for Spoken Language Translation","date":"2024-11-27","arxiv_id":"2411.18294","repositories_listed":0,"syntology":null},{"url":null,"slug":"amps-asr-with-multimodal-paraphrase","title":"AMPS: ASR with Multimodal Paraphrase Supervision","date":"2024-11-27","arxiv_id":"2411.18368","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-in-machine-speech-chain","title":"Continual Learning in Machine Speech Chain Using Gradient Episodic Memory","date":"2024-11-27","arxiv_id":"2411.18320","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-learn-a-new-language-an-efficient","title":"How to Learn a New Language? An Efficient Solution for Self-Supervised Learning Models Unseen Languages Adaption in Low-Resource Scenario","date":"2024-11-27","arxiv_id":"2411.18217","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-transformer-an-explainable-end","title":"Disentangled-Transformer: An Explainable End-to-End Automatic Speech Recognition Model with Speech Content-Context Separation","date":"2024-11-26","arxiv_id":"2411.17846","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-code-switching-asr-leveraging-non","title":"Enhancing Code-Switching ASR Leveraging Non-Peaky CTC Loss and Deep Language Posterior Injection","date":"2024-11-26","arxiv_id":"2412.08651","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-maximum-likelihood-training-for","title":"Towards Maximum Likelihood Training for Transducer-based Streaming Speech Recognition","date":"2024-11-26","arxiv_id":"2411.17537","repositories_listed":0,"syntology":null},{"url":"/paper/high-precision-medical-speech-recognition","slug":"high-precision-medical-speech-recognition","title":"High-precision medical speech recognition through synthetic data and semantic correction: UNITED-MEDASR","date":"2024-11-24","arxiv_id":"2412.00055","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-nlu-with-babylon-a-case-study-in","title":"Transforming NLU with Babylon: A Case Study in Development of Real-time, Edge-Efficient, Multi-Intent Translation System for Automated Drive-Thru Ordering","date":"2024-11-22","arxiv_id":"2411.15372","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiny-align-bridging-automatic-speech","title":"Tiny-Align: Bridging Automatic Speech Recognition and Large Language Model on the Edge","date":"2024-11-21","arxiv_id":"2411.13766","repositories_listed":0,"syntology":null},{"url":null,"slug":"cafe-a-novel-code-switching-dataset-for","title":"CAFE A Novel Code switching Dataset for Algerian Dialect French and English","date":"2024-11-20","arxiv_id":"2411.13424","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-statistical-methods-to-pre-trained","title":"From Statistical Methods to Pre-Trained Models; A Survey on Automatic Speech Recognition for Resource Scarce Urdu Language","date":"2024-11-20","arxiv_id":"2411.14493","repositories_listed":0,"syntology":null},{"url":null,"slug":"hard-synth-synthesizing-diverse-hard-samples","title":"Hard-Synth: Synthesizing Diverse Hard Samples for ASR using Zero-Shot TTS and LLM","date":"2024-11-20","arxiv_id":"2411.13159","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisper-finetuning-on-nepali-language","title":"Whisper Finetuning on Nepali Language","date":"2024-11-19","arxiv_id":"2411.12587","repositories_listed":0,"syntology":null},{"url":null,"slug":"everyone-deserves-their-voice-to-be-heard","title":"Everyone deserves their voice to be heard: Analyzing Predictive Gender Bias in ASR Models Applied to Dutch Speech Data","date":"2024-11-14","arxiv_id":"2411.09431","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-adversarial-attacks-against-asr","title":"Transferable Adversarial Attacks against ASR","date":"2024-11-14","arxiv_id":"2411.09220","repositories_listed":0,"syntology":null},{"url":null,"slug":"dcf-ds-deep-cascade-fusion-of-diarization-and","title":"DCF-DS: Deep Cascade Fusion of Diarization and Separation for Speech Recognition under Realistic Single-Channel Conditions","date":"2024-11-11","arxiv_id":"2411.06667","repositories_listed":0,"syntology":null},{"url":null,"slug":"multistage-fine-tuning-strategies-for","title":"Multistage Fine-tuning Strategies for Automatic Speech Recognition in Low-resource Languages","date":"2024-11-07","arxiv_id":"2411.04573","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-aac-software-for-dysarthric","title":"Enhancing AAC Software for Dysarthric Speakers in e-Health Settings: An Evaluation Using TORGO","date":"2024-11-01","arxiv_id":"2411.00980","repositories_listed":0,"syntology":null},{"url":null,"slug":"run-time-adaptation-of-neural-beamforming-for","title":"Run-Time Adaptation of Neural Beamforming for Robust Speech Dereverberation and Denoising","date":"2024-10-30","arxiv_id":"2410.22805","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-based-emotion-recognition","title":"Improving Speech-based Emotion Recognition with Contextual Utterance Analysis and LLMs","date":"2024-10-27","arxiv_id":"2410.20334","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-speech-large-language-models","title":"A Survey on Speech Large Language Models","date":"2024-10-24","arxiv_id":"2410.18908","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-improving-automatic-speech","title":"Evaluating and Improving Automatic Speech Recognition Systems for Korean Meteorological Experts","date":"2024-10-24","arxiv_id":"2410.18444","repositories_listed":0,"syntology":null},{"url":null,"slug":"elaichi-enhancing-low-resource-tts-by","title":"ELAICHI: Enhancing Low-resource TTS by Addressing Infrequent and Low-frequency Character Bigrams","date":"2024-10-23","arxiv_id":"2410.17901","repositories_listed":0,"syntology":null}],"record_sha256":"edf64b18ed6b460dab75993f89a66eff9ea920372f2ab95ed9c3cf0b4deaadab","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}