{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/17","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":65,"rows_per_page":100,"rows":[1601,1700],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/16","next":"/task/speech-recognition/papers/18","papers":[{"url":null,"slug":"classification-error-bound-for-low-bayes","title":"Classification Error Bound for Low Bayes Error Conditions in Machine Learning","date":"2025-01-27","arxiv_id":"2501.15977","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-target-speaker-speech-recognition","title":"End-to-End Target Speaker Speech Recognition Using Context-Aware Attention Mechanisms for Challenging Enrollment Scenario","date":"2025-01-26","arxiv_id":"2501.15466","repositories_listed":0,"syntology":null},{"url":null,"slug":"seal-speech-embedding-alignment-learning-for","title":"SEAL: Speech Embedding Alignment Learning for Speech Large Language Model with Retrieval-Augmented Generation","date":"2025-01-26","arxiv_id":"2502.02603","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-cross-etiology-and-speaker-independent","title":"Robust Cross-Etiology and Speaker-Independent Dysarthric Speech Recognition","date":"2025-01-25","arxiv_id":"2501.14994","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multicultural-medical-assistant-can-llms","title":"The Multicultural Medical Assistant: Can LLMs Improve Medical ASR Errors Across Borders?","date":"2025-01-25","arxiv_id":"2501.15310","repositories_listed":0,"syntology":null},{"url":null,"slug":"locoml-a-framework-for-real-world-ml","title":"LoCoML: A Framework for Real-World ML Inference Pipelines","date":"2025-01-24","arxiv_id":"2501.14165","repositories_listed":0,"syntology":null},{"url":null,"slug":"dq-data2vec-decoupling-quantization-for","title":"DQ-Data2vec: Decoupling Quantization for Multilingual Speech Recognition","date":"2025-01-23","arxiv_id":"2501.13497","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-persian-lip-reading-in-surena-v","title":"Integrating Persian Lip Reading in Surena-V Humanoid Robot for Human-Robot Interaction","date":"2025-01-23","arxiv_id":"2501.13996","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-a-posteriori-speech-presence","title":"Learning-based A Posteriori Speech Presence Probability Estimation and Applications","date":"2025-01-23","arxiv_id":"2501.13642","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-compact-phrasal-rewrites-with","title":"Predicting Compact Phrasal Rewrites with Large Language Models for ASR Post Editing","date":"2025-01-23","arxiv_id":"2501.13831","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-an-inclusive-educational","title":"Development of an Inclusive Educational Platform Using Open Technologies and Machine Learning: A Case Study on Accessibility Enhancement","date":"2025-01-22","arxiv_id":"2503.15501","repositories_listed":0,"syntology":null},{"url":"/paper/let-ssms-be-convnets-state-space-modeling","slug":"let-ssms-be-convnets-state-space-modeling","title":"Let SSMs be ConvNets: State-space Modeling with Optimal Tensor Contractions","date":"2025-01-22","arxiv_id":"2501.13230","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-domain-adaptation-framework-for-speech","title":"A Domain Adaptation Framework for Speech Recognition Systems with Only Synthetic data","date":"2025-01-21","arxiv_id":"2501.12501","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ai-and-large-language-models-in","title":"Generative AI and Large Language Models in Language Preservation: Opportunities and Challenges","date":"2025-01-20","arxiv_id":"2501.11496","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-whisper-asr-hallucinations","title":"Investigation of Whisper ASR Hallucinations Induced by Non-Speech Audio","date":"2025-01-20","arxiv_id":"2501.11378","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-neural-spoken-language-recognition","title":"Enhancing Neural Spoken Language Recognition: An Exploration with Multilingual Datasets","date":"2025-01-19","arxiv_id":"2501.11065","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmark-of-french-asr-systems-based-on","title":"A Benchmark of French ASR Systems Based on Error Severity","date":"2025-01-18","arxiv_id":"2501.10879","repositories_listed":0,"syntology":null},{"url":null,"slug":"gec-rag-improving-generative-error-correction","title":"GEC-RAG: Improving Generative Error Correction via Retrieval-Augmented Generation for Automatic Speech Recognition Systems","date":"2025-01-18","arxiv_id":"2501.10734","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-sanskrit","title":"Automatic Speech Recognition for Sanskrit with Transfer Learning","date":"2025-01-17","arxiv_id":"2501.10024","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-rhythm-and-voice-conversion-of","title":"Unsupervised Rhythm and Voice Conversion of Dysarthric to Healthy Speech for ASR","date":"2025-01-17","arxiv_id":"2501.10256","repositories_listed":0,"syntology":null},{"url":null,"slug":"delayed-fusion-integrating-large-language","title":"Delayed Fusion: Integrating Large Language Models into First-Pass Decoding in End-to-end Speech Recognition","date":"2025-01-16","arxiv_id":"2501.09258","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-non-autoregressive-model-for-joint-stt-and","title":"A Non-autoregressive Model for Joint STT and TTS","date":"2025-01-15","arxiv_id":"2501.09104","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-whisper-for-regional-dialects","title":"Adapting Whisper for Regional Dialects: Enhancing Public Services for Vulnerable Populations in the United Kingdom","date":"2025-01-15","arxiv_id":"2501.08502","repositories_listed":0,"syntology":null},{"url":null,"slug":"persoda-personalized-data-augmentation","title":"persoDA: Personalized Data Augmentation for Personalized ASR","date":"2025-01-15","arxiv_id":"2501.09113","repositories_listed":0,"syntology":null},{"url":null,"slug":"loudspeaker-beamforming-to-enhance-speech","title":"Loudspeaker Beamforming to Enhance Speech Recognition Performance of Voice Driven Applications","date":"2025-01-14","arxiv_id":"2501.08104","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-cot-exploring-chain-of-thought","title":"Audio-CoT: Exploring Chain-of-Thought Reasoning in Large Audio Language Model","date":"2025-01-13","arxiv_id":"2501.07246","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-spoken-italian-datasets-and","title":"A Survey on Spoken Italian Datasets and Corpora","date":"2025-01-11","arxiv_id":"2501.06557","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-for-automatically","title":"Speech Recognition for Automatically Assessing Afrikaans and isiXhosa Preschool Oral Narratives","date":"2025-01-11","arxiv_id":"2501.06478","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-rotary-position-embeddings-for","title":"Benchmarking Rotary Position Embeddings for Automatic Speech Recognition","date":"2025-01-10","arxiv_id":"2501.06051","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-asr-error-handling-with-llms","title":"Contextual ASR Error Handling with LLMs Augmentation for Goal-Oriented Conversational AI","date":"2025-01-10","arxiv_id":"2501.06129","repositories_listed":0,"syntology":null},{"url":null,"slug":"tts-transducer-end-to-end-speech-synthesis","title":"TTS-Transducer: End-to-End Speech Synthesis with Neural Transducer","date":"2025-01-10","arxiv_id":"2501.06320","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-2-tf-robust-all-neural-text","title":"Universal-2-TF: Robust All-Neural Text Formatting for ASR","date":"2025-01-10","arxiv_id":"2501.05948","repositories_listed":0,"syntology":null},{"url":null,"slug":"lipgen-viseme-guided-lip-video-generation-for","title":"LipGen: Viseme-Guided Lip Video Generation for Enhancing Visual Speech Recognition","date":"2025-01-08","arxiv_id":"2501.04204","repositories_listed":0,"syntology":null},{"url":null,"slug":"methods-to-increase-the-amount-of-data-for","title":"Methods to Increase the Amount of Data for Speech Recognition for Low Resource Languages","date":"2025-01-08","arxiv_id":"2501.14788","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-for-pathological-speech-a","title":"Deep Learning for Pathological Speech: A Survey","date":"2025-01-07","arxiv_id":"2501.03536","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-generalizable-speech-marker-for","title":"Towards a Generalizable Speech Marker for Parkinson's Disease Diagnosis","date":"2025-01-07","arxiv_id":"2501.03581","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-speaker-embedding-free-target","title":"Universal Speaker Embedding Free Target Speaker Extraction and Personal Voice Activity Detection","date":"2025-01-07","arxiv_id":"2501.03612","repositories_listed":0,"syntology":null},{"url":"/paper/samba-asr-state-of-the-art-speech-recognition","slug":"samba-asr-state-of-the-art-speech-recognition","title":"Samba-ASR: State-Of-The-Art Speech Recognition Leveraging Structured State-Space Models","date":"2025-01-06","arxiv_id":"2501.02832","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-transducer-based-spoken-language","title":"Improving Transducer-Based Spoken Language Understanding with Self-Conditioned CTC and Knowledge Transfer","date":"2025-01-03","arxiv_id":"2501.01936","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-text-pronunciation-correlation","title":"Automatic Text Pronunciation Correlation Generation and Application for Contextual Biasing","date":"2025-01-01","arxiv_id":"2501.00804","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-through-the-spike-spike-window","title":"Breaking Through the Spike: Spike Window Decoding for Accelerated and Precise Automatic Speech Recognition","date":"2025-01-01","arxiv_id":"2501.03257","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-dialogue-management-survey","title":"Incremental Dialogue Management: Survey, Discussion, and Implications for HRI","date":"2025-01-01","arxiv_id":"2501.00953","repositories_listed":0,"syntology":null},{"url":null,"slug":"livecc-learning-video-llm-with-streaming","title":"LiveCC: Learning Video LLM with Streaming Speech Transcription at Scale","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fotheidil-an-automatic-transcription-system","title":"Fotheidil: an Automatic Transcription System for the Irish Language","date":"2024-12-31","arxiv_id":"2501.00509","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisper-turns-stronger-augmenting-wav2vec-2-0","title":"Whisper Turns Stronger: Augmenting Wav2Vec 2.0 for Superior ASR in Low-Resource Languages","date":"2024-12-31","arxiv_id":"2501.00425","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-whisper-s-accuracy-and-speed-for","title":"Enhancing Whisper's Accuracy and Speed for Indian Languages through Prompt-Tuning and Tokenization","date":"2024-12-27","arxiv_id":"2412.19785","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-single-asr-model-that-generalizes","title":"Towards a Single ASR Model That Generalizes to Disordered Speech","date":"2024-12-26","arxiv_id":"2412.19315","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-with-llms-adapted-to","title":"Speech Recognition With LLMs Adapted to Disordered Speech Using Reinforcement Learning","date":"2024-12-25","arxiv_id":"2501.00039","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-speaker-deficiency-adaptation-of","title":"Structured Speaker-Deficiency Adaptation of Foundation Models for Dysarthric and Elderly Speech Recognition","date":"2024-12-25","arxiv_id":"2412.18832","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-resource-speech-translation-and","title":"Zero-resource Speech Translation and Recognition with LLMs","date":"2024-12-24","arxiv_id":"2412.18566","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-in-proteomics-informatics","title":"Deep Learning in Proteomics Informatics: Applications, Challenges, and Future Directions","date":"2024-12-23","arxiv_id":"2412.17349","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-prosodic-signatures-via-speech","title":"Investigating Prosodic Signatures via Speech Pre-Trained Models for Audio Deepfake Source Attribution","date":"2024-12-23","arxiv_id":"2412.17796","repositories_listed":0,"syntology":null},{"url":null,"slug":"trading-devil-rl-backdoor-attack-via-stock","title":"Trading Devil RL: Backdoor attack via Stock market, Bayesian Optimization and Reinforcement Learning","date":"2024-12-23","arxiv_id":"2412.17908","repositories_listed":0,"syntology":null},{"url":null,"slug":"ume-upcycling-mixture-of-experts-for-scalable","title":"UME: Upcycling Mixture-of-Experts for Scalable and Efficient Automatic Speech Recognition","date":"2024-12-23","arxiv_id":"2412.17507","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncovering-the-visual-contribution-in-audio","title":"Uncovering the Visual Contribution in Audio-Visual Speech Recognition","date":"2024-12-22","arxiv_id":"2412.17129","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-whisper-for-code-switching-through","title":"Adapting Whisper for Code-Switching through Encoding Refining and Language-Aware Decoding","date":"2024-12-21","arxiv_id":"2412.16507","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multilingual-asr-for-unseen","title":"Enhancing Multilingual ASR for Unseen Languages via Language Embedding Modeling","date":"2024-12-21","arxiv_id":"2412.16474","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-retrieval-augmented-generation-without","title":"Speech Retrieval-Augmented Generation without Automatic Speech Recognition","date":"2024-12-21","arxiv_id":"2412.16500","repositories_listed":0,"syntology":null},{"url":null,"slug":"transducer-llama-integrating-llms-into","title":"Transducer-Llama: Integrating LLMs into Streamable Transducer-based Speech Recognition","date":"2024-12-21","arxiv_id":"2412.16464","repositories_listed":0,"syntology":null},{"url":null,"slug":"touchasp-elastic-automatic-speech-perception","title":"TouchASP: Elastic Automatic Speech Perception that Everyone Can Touch","date":"2024-12-20","arxiv_id":"2412.15622","repositories_listed":0,"syntology":null},{"url":null,"slug":"lama-ut-language-agnostic-multilingual-asr","title":"LAMA-UT: Language Agnostic Multilingual ASR through Orthography Unification and Language-Specific Transliteration","date":"2024-12-19","arxiv_id":"2412.15299","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribing-and-translating-fast-and-slow","title":"Transcribing and Translating, Fast and Slow: Joint Speech Translation and Recognition","date":"2024-12-19","arxiv_id":"2412.15415","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-improve-challenge-2025-tasks-and","title":"Speak & Improve Challenge 2025: Tasks and Baseline Systems","date":"2024-12-16","arxiv_id":"2412.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-improve-corpus-2025-an-l2-english","title":"Speak & Improve Corpus 2025: an L2 English Speech Corpus for Language Assessment and Feedback","date":"2024-12-16","arxiv_id":"2412.11986","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-speech-foundation-model-for","title":"MERaLiON-SpeechEncoder: Towards a Speech Foundation Model for Singapore and Beyond","date":"2024-12-16","arxiv_id":"2412.11538","repositories_listed":0,"syntology":null},{"url":null,"slug":"transliterated-zero-shot-domain-adaptation","title":"Transliterated Zero-Shot Domain Adaptation for Automatic Speech Recognition","date":"2024-12-15","arxiv_id":"2412.11185","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-recognition-of-persian-isolated-digits","title":"Robust Persian Digit Recognition in Noisy Environments Using Hybrid CNN-BiGRU Model","date":"2024-12-14","arxiv_id":"2412.10857","repositories_listed":0,"syntology":null},{"url":null,"slug":"meralion-audiollm-technical-report","title":"MERaLiON-AudioLLM: Bridging Audio and Language with Large Language Models","date":"2024-12-13","arxiv_id":"2412.09818","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilevel-joint-unsupervised-and-supervised","title":"Bilevel Joint Unsupervised and Supervised Training for Automatic Speech Recognition","date":"2024-12-11","arxiv_id":"2412.08548","repositories_listed":0,"syntology":null},{"url":null,"slug":"greek2mathtex-a-greek-speech-to-text","title":"Greek2MathTex: A Greek Speech-to-Text Framework for LaTeX Equations Generation","date":"2024-12-11","arxiv_id":"2412.12167","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-agnostic-evaluation-of-asr-using","title":"Style-agnostic evaluation of ASR using multiple reference transcripts","date":"2024-12-10","arxiv_id":"2412.07937","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-text-adaptation-for-llm-based-asr","title":"Effective Text Adaptation for LLM-based ASR through Soft Prompt Fine-Tuning","date":"2024-12-09","arxiv_id":"2412.06967","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-machine-learning-model-for-inner","title":"Ensemble Machine Learning Model for Inner Speech Recognition: A Subject-Specific Investigation","date":"2024-12-09","arxiv_id":"2412.17824","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-transfer-learning-from-swahili","title":"Harnessing Transfer Learning from Swahili: Advancing Solutions for Comorian Dialects","date":"2024-12-09","arxiv_id":"2412.12143","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-prompt-learning-and-pause-encoding","title":"Leveraging Prompt Learning and Pause Encoding for Alzheimer's Disease Detection","date":"2024-12-09","arxiv_id":"2412.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-all-errors-are-equal-investigation-of","title":"Not All Errors Are Equal: Investigation of Speech Recognition Errors in Alzheimer's Disease Detection","date":"2024-12-09","arxiv_id":"2412.06332","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-dropout-for-pruning-conformers","title":"Adaptive Dropout for Pruning Conformers","date":"2024-12-06","arxiv_id":"2412.04836","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-audio-query-handling-system","title":"Comprehensive Audio Query Handling System with Integrated Expert Models and Contextual Understanding","date":"2024-12-05","arxiv_id":"2412.03980","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-based-feature-extraction","title":"Speech Recognition-based Feature Extraction for Enhanced Automatic Severity Classification in Dysarthric Speech","date":"2024-12-05","arxiv_id":"2412.03784","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-ec-benchmark-evaluating-large-language","title":"ASR-EC Benchmark: Evaluating Large Language Models on Chinese ASR Error Correction","date":"2024-12-04","arxiv_id":"2412.03075","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-llm-based-asr-and","title":"A Comparative Study of LLM-based ASR and Whisper in Low Resource and Code Switching Scenario","date":"2024-12-01","arxiv_id":"2412.00721","repositories_listed":0,"syntology":null},{"url":null,"slug":"late-fusion-ensembles-for-speech-recognition","title":"Late fusion ensembles for speech recognition on diverse input audio representations","date":"2024-12-01","arxiv_id":"2412.01861","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-the-deaf-and-hard-of-hearing","title":"Empowering the Deaf and Hard of Hearing Community: Enhancing Video Captions Using Large Language Models","date":"2024-11-30","arxiv_id":"2412.00342","repositories_listed":0,"syntology":null},{"url":"/paper/areeg-words-dataset-for-envisioned-speech","slug":"areeg-words-dataset-for-envisioned-speech","title":"ArEEG_Words: Dataset for Envisioned Speech Recognition using EEG for Arabic Words","date":"2024-11-28","arxiv_id":"2411.18888","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-pre-trained-models-for-spoken","title":"Aligning Pre-trained Models for Spoken Language Translation","date":"2024-11-27","arxiv_id":"2411.18294","repositories_listed":0,"syntology":null},{"url":null,"slug":"amps-asr-with-multimodal-paraphrase","title":"AMPS: ASR with Multimodal Paraphrase Supervision","date":"2024-11-27","arxiv_id":"2411.18368","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-in-machine-speech-chain","title":"Continual Learning in Machine Speech Chain Using Gradient Episodic Memory","date":"2024-11-27","arxiv_id":"2411.18320","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-learn-a-new-language-an-efficient","title":"How to Learn a New Language? An Efficient Solution for Self-Supervised Learning Models Unseen Languages Adaption in Low-Resource Scenario","date":"2024-11-27","arxiv_id":"2411.18217","repositories_listed":0,"syntology":null},{"url":null,"slug":"msa-asr-efficient-multilingual-speaker","title":"MSA-ASR: Efficient Multilingual Speaker Attribution with frozen ASR Models","date":"2024-11-27","arxiv_id":"2411.18152","repositories_listed":0,"syntology":null},{"url":null,"slug":"salmonn-omni-a-codec-free-llm-for-full-duplex","title":"SALMONN-omni: A Codec-free LLM for Full-duplex Speech Understanding and Generation","date":"2024-11-27","arxiv_id":"2411.18138","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-transformer-an-explainable-end","title":"Disentangled-Transformer: An Explainable End-to-End Automatic Speech Recognition Model with Speech Content-Context Separation","date":"2024-11-26","arxiv_id":"2411.17846","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-code-switching-asr-leveraging-non","title":"Enhancing Code-Switching ASR Leveraging Non-Peaky CTC Loss and Deep Language Posterior Injection","date":"2024-11-26","arxiv_id":"2412.08651","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-maximum-likelihood-training-for","title":"Towards Maximum Likelihood Training for Transducer-based Streaming Speech Recognition","date":"2024-11-26","arxiv_id":"2411.17537","repositories_listed":0,"syntology":null},{"url":"/paper/high-precision-medical-speech-recognition","slug":"high-precision-medical-speech-recognition","title":"High-precision medical speech recognition through synthetic data and semantic correction: UNITED-MEDASR","date":"2024-11-24","arxiv_id":"2412.00055","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-nlu-with-babylon-a-case-study-in","title":"Transforming NLU with Babylon: A Case Study in Development of Real-time, Edge-Efficient, Multi-Intent Translation System for Automated Drive-Thru Ordering","date":"2024-11-22","arxiv_id":"2411.15372","repositories_listed":0,"syntology":null},{"url":null,"slug":"tskips-efficiency-through-explicit-temporal","title":"TSkips: Efficiency Through Explicit Temporal Delay Connections in Spiking Neural Networks","date":"2024-11-22","arxiv_id":"2411.16711","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiny-align-bridging-automatic-speech","title":"Tiny-Align: Bridging Automatic Speech Recognition and Large Language Model on the Edge","date":"2024-11-21","arxiv_id":"2411.13766","repositories_listed":0,"syntology":null},{"url":null,"slug":"cafe-a-novel-code-switching-dataset-for","title":"CAFE A Novel Code switching Dataset for Algerian Dialect French and English","date":"2024-11-20","arxiv_id":"2411.13424","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-statistical-methods-to-pre-trained","title":"From Statistical Methods to Pre-Trained Models; A Survey on Automatic Speech Recognition for Resource Scarce Urdu Language","date":"2024-11-20","arxiv_id":"2411.14493","repositories_listed":0,"syntology":null},{"url":null,"slug":"hard-synth-synthesizing-diverse-hard-samples","title":"Hard-Synth: Synthesizing Diverse Hard Samples for ASR using Zero-Shot TTS and LLM","date":"2024-11-20","arxiv_id":"2411.13159","repositories_listed":0,"syntology":null}],"record_sha256":"4cd993f521b58ceb51626800d0a4549ef16773e125a9c6a99be30681cb6974d4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}