{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/15","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":58,"rows_per_page":100,"rows":[1401,1500],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/14","next":"/task/speech-recognition-1/papers/16","papers":[{"url":null,"slug":"lipdiffuser-lip-to-speech-generation-with","title":"LipDiffuser: Lip-to-Speech Generation with Conditional Diffusion Models","date":"2025-05-16","arxiv_id":"2505.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"inclusivity-of-ai-speech-in-healthcare-a","title":"Inclusivity of AI Speech in Healthcare: A Decade Look Back","date":"2025-05-15","arxiv_id":"2505.10596","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantized-approximate-signal-processing-qasp","title":"Quantized Approximate Signal Processing (QASP): Towards Homomorphic Encryption for audio","date":"2025-05-15","arxiv_id":"2505.10500","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-simulation-on-the-dynamics-of-auditory","title":"Full simulation on the dynamics of auditory synaptic fusion: Strong clustering of calcium channel might be the origin of the coherent release in the auditory hair cells","date":"2025-05-12","arxiv_id":"2505.07273","repositories_listed":0,"syntology":null},{"url":null,"slug":"remote-rowhammer-attack-using-adversarial","title":"Remote Rowhammer Attack using Adversarial Observations on Federated Learning Clients","date":"2025-05-09","arxiv_id":"2505.06335","repositories_listed":0,"syntology":null},{"url":null,"slug":"teochew-wild-the-first-in-the-wild-teochew","title":"Teochew-Wild: The First In-the-wild Teochew Dataset with Orthographic Annotations","date":"2025-05-08","arxiv_id":"2505.05056","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-speech-recognition-with-schrodinger","title":"Robust Speech Recognition with Schrödinger Bridge-Based Speech Enhancement","date":"2025-05-07","arxiv_id":"2505.04237","repositories_listed":0,"syntology":null},{"url":null,"slug":"swinlip-an-efficient-visual-speech-encoder","title":"SwinLip: An Efficient Visual Speech Encoder for Lip Reading Using Swin Transformer","date":"2025-05-07","arxiv_id":"2505.04394","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-of-automatic-speech-recognition-in","title":"Fairness of Automatic Speech Recognition in Cleft Lip and Palate Speech","date":"2025-05-06","arxiv_id":"2505.03697","repositories_listed":0,"syntology":null},{"url":null,"slug":"sepalm-audio-language-models-are-error","title":"SepALM: Audio Language Models Are Error Correctors for Robust Speech Separation","date":"2025-05-06","arxiv_id":"2505.03273","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-synergistic-framework-of-nonlinear-acoustic","title":"A Synergistic Framework of Nonlinear Acoustic Computing and Reinforcement Learning for Real-World Human-Robot Interaction","date":"2025-05-04","arxiv_id":"2505.01998","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-based-deep-residual","title":"Transfer Learning-Based Deep Residual Learning for Speech Recognition in Clean and Noisy Environments","date":"2025-05-02","arxiv_id":"2505.01632","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-enhanced-few-shot-prompting-for","title":"Retrieval-Enhanced Few-Shot Prompting for Speech Event Extraction","date":"2025-04-30","arxiv_id":"2504.21372","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-and-evaluation-of-a-deep-learning-1","title":"Development and evaluation of a deep learning algorithm for German word recognition from lip movements","date":"2025-04-22","arxiv_id":"2504.15792","repositories_listed":0,"syntology":null},{"url":null,"slug":"chinese-lips-a-chinese-audio-visual-speech","title":"Chinese-LiPS: A Chinese audio-visual speech recognition dataset with Lip-reading and Presentation Slides","date":"2025-04-21","arxiv_id":"2504.15066","repositories_listed":0,"syntology":null},{"url":null,"slug":"stablequant-layer-adaptive-post-training","title":"StableQuant: Layer Adaptive Post-Training Quantization for Speech Foundation Models","date":"2025-04-21","arxiv_id":"2504.14915","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-to-articulatory-inversion-of-speech","title":"Acoustic to Articulatory Inversion of Speech; Data Driven Approaches, Challenges, Applications, and Future Scope","date":"2025-04-17","arxiv_id":"2504.13308","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-arabic-speech-recognition-through","title":"Advancing Arabic Speech Recognition Through Large-Scale Weakly Supervised Learning","date":"2025-04-16","arxiv_id":"2504.12254","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-audio-processing-with-large-language","title":"Spatial Audio Processing with Large Language Model on Wearable Devices","date":"2025-04-11","arxiv_id":"2504.08907","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-speech-to-summary-a-comprehensive-survey","title":"Summarizing Speech: A Comprehensive Survey","date":"2025-04-10","arxiv_id":"2504.08024","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-aware-speech-recognition-for-noisy","title":"Visual-Aware Speech Recognition for Noisy Scenarios","date":"2025-04-09","arxiv_id":"2504.07229","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-human-digital-twin-architecture-for","title":"A Human Digital Twin Architecture for Knowledge-based Interactions and Context-Aware Conversations","date":"2025-04-04","arxiv_id":"2504.03147","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-intelligence-for-wildlife-conservation","title":"Edge Intelligence for Wildlife Conservation: Real-Time Hornbill Call Classification Using TinyML","date":"2025-04-03","arxiv_id":"2504.12272","repositories_listed":0,"syntology":null},{"url":null,"slug":"linto-audio-and-textual-datasets-to-train-and","title":"LinTO Audio and Textual Datasets to Train and Evaluate Automatic Speech Recognition in Tunisian Arabic Dialect","date":"2025-04-03","arxiv_id":"2504.02604","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-correction-for-full-text-speech","title":"Chain of Correction for Full-text Speech Recognition with Large Language Models","date":"2025-04-02","arxiv_id":"2504.01519","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-auditory-cognition-via-test-time","title":"Scaling Auditory Cognition via Test-Time Compute in Audio Language Models","date":"2025-03-30","arxiv_id":"2503.23395","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-code-switched-synthetic-data","title":"The Impact of Code-switched Synthetic Data Quality is Task Dependent: Insights from MT and ASR","date":"2025-03-30","arxiv_id":"2503.23576","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-71-2-m-w-speech-recognition-accelerator","title":"A 71.2-$μ$W Speech Recognition Accelerator with Recurrent Spiking Neural Network","date":"2025-03-27","arxiv_id":"2503.21337","repositories_listed":0,"syntology":null},{"url":null,"slug":"vallr-visual-asr-language-model-for-lip","title":"VALLR: Visual ASR Language Model for Lip Reading","date":"2025-03-27","arxiv_id":"2503.21408","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-first-order-optimization-on-the","title":"Efficient First-Order Optimization on the Pareto Set for Multi-Objective Learning under Preference Guidance","date":"2025-03-26","arxiv_id":"2504.02854","repositories_listed":0,"syntology":null},{"url":null,"slug":"finaudio-a-benchmark-for-audio-large-language","title":"FinAudio: A Benchmark for Audio Large Language Models in Financial Applications","date":"2025-03-26","arxiv_id":"2503.20990","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-accuracy-using","title":"Improving Speech Recognition Accuracy Using Custom Language Models with the Vosk Toolkit","date":"2025-03-26","arxiv_id":"2503.21025","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-the-transferability-of-audio","title":"Boosting the Transferability of Audio Adversarial Examples with Acoustic Representation Optimization","date":"2025-03-25","arxiv_id":"2503.19591","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-metric-meta-evaluation-by","title":"Contextual Metric Meta-Evaluation by Measuring Local Metric Accuracy","date":"2025-03-25","arxiv_id":"2503.19828","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-calibrated-affective-speech-recognition","title":"Coverage-Guaranteed Speech Emotion Recognition via Calibrated Uncertainty-Adaptive Prediction Sets","date":"2025-03-24","arxiv_id":"2503.22712","repositories_listed":0,"syntology":null},{"url":null,"slug":"whispering-in-amharic-fine-tuning-whisper-for","title":"Whispering in Amharic: Fine-tuning Whisper for Low-resource Language","date":"2025-03-24","arxiv_id":"2503.18485","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-structured-state-space-sequence","title":"From S4 to Mamba: A Comprehensive Survey on Structured State Space Models","date":"2025-03-22","arxiv_id":"2503.18970","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-voice-is-your-voice-supporting-self","title":"Your voice is your voice: Supporting Self-expression through Speech Generation and LLMs in Augmented and Alternative Communication","date":"2025-03-21","arxiv_id":"2503.17479","repositories_listed":0,"syntology":null},{"url":null,"slug":"seniortalk-a-chinese-conversation-dataset","title":"SeniorTalk: A Chinese Conversation Dataset with Rich Annotations for Super-Aged Seniors","date":"2025-03-20","arxiv_id":"2503.16578","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-architectural","title":"A Comprehensive Survey on Architectural Advances in Deep CNNs: Challenges, Applications, and Emerging Research Directions","date":"2025-03-19","arxiv_id":"2503.16546","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-asr-confidence-scores-for","title":"Evaluating ASR Confidence Scores for Automated Error Detection in User-Assisted Correction Interfaces","date":"2025-03-19","arxiv_id":"2503.15124","repositories_listed":0,"syntology":null},{"url":null,"slug":"halving-transcription-time-a-fast-user","title":"Halving transcription time: A fast, user-friendly and GDPR-compliant workflow to create AI-assisted transcripts for content analysis","date":"2025-03-17","arxiv_id":"2503.13031","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-aviation-communication","title":"Enhancing Aviation Communication Transcription: Fine-Tuning Distil-Whisper with LoRA","date":"2025-03-13","arxiv_id":"2503.22692","repositories_listed":0,"syntology":null},{"url":null,"slug":"everything-can-be-described-in-words-a-simple","title":"Everything Can Be Described in Words: A Simple Unified Multi-Modal Framework with Semantic and Temporal Alignment","date":"2025-03-12","arxiv_id":"2503.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"proceedings-of-the-isca-itg-workshop-on","title":"Proceedings of the ISCA/ITG Workshop on Diversity in Large Speech and Language Models","date":"2025-03-12","arxiv_id":"2503.10298","repositories_listed":0,"syntology":null},{"url":null,"slug":"valsub-subsampling-validation-data-to","title":"ValSub: Subsampling Validation Data to Mitigate Forgetting during ASR Personalization","date":"2025-03-12","arxiv_id":"2503.09906","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exhaustive-evaluation-of-tts-and-vc-based","title":"An Exhaustive Evaluation of TTS- and VC-based Data Augmentation for ASR","date":"2025-03-11","arxiv_id":"2503.08954","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-non-native","title":"Automatic Speech Recognition for Non-Native English: Accuracy and Disfluency Handling","date":"2025-03-10","arxiv_id":"2503.06924","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-english-asr-model-with-regional","title":"Building English ASR model with regional language support","date":"2025-03-10","arxiv_id":"2503.07522","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-audio-visual-speech-recognition-via","title":"Adaptive Audio-Visual Speech Recognition via Matryoshka-Based Multimodal LLMs","date":"2025-03-09","arxiv_id":"2503.06362","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-causal-inference-approach-for-quantifying","title":"A Causal Inference Approach for Quantifying Research Impact","date":"2025-03-07","arxiv_id":"2503.13485","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-voice-to-safety-language-ai-powered","title":"From Voice to Safety: Language AI Powered Pilot-ATC Communication Understanding for Airport Surface Movement Collision Risk Assessment","date":"2025-03-06","arxiv_id":"2503.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-models-for-phoneme","title":"Self-Supervised Models for Phoneme Recognition: Applications in Children's Speech for Reading Learning","date":"2025-03-06","arxiv_id":"2503.04710","repositories_listed":0,"syntology":null},{"url":null,"slug":"qieemo-speech-is-all-you-need-in-the-emotion","title":"Qieemo: Speech Is All You Need in the Emotion Recognition in Conversations","date":"2025-03-05","arxiv_id":"2503.22687","repositories_listed":0,"syntology":null},{"url":null,"slug":"cordic-is-all-you-need","title":"CORDIC Is All You Need","date":"2025-03-04","arxiv_id":"2503.11685","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-speech-to-speech-translation-a-review","title":"Direct Speech to Speech Translation: A Review","date":"2025-03-03","arxiv_id":"2503.04799","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-whisper-for-inclusive-prosodic","title":"Fine-Tuning Whisper for Inclusive Prosodic Stress Analysis","date":"2025-03-03","arxiv_id":"2503.02907","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniwav-towards-unified-pre-training-for","title":"UniWav: Towards Unified Pre-training for Speech Representation Learning and Generation","date":"2025-03-02","arxiv_id":"2503.00733","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-biases-while-embracing","title":"Unveiling Biases while Embracing Sustainability: Assessing the Dual Challenges of Automatic Speech Recognition Systems","date":"2025-03-02","arxiv_id":"2503.00907","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-automatic-speech-recognition-for","title":"Adapting Automatic Speech Recognition for Accented Air Traffic Control Communications","date":"2025-02-27","arxiv_id":"2502.20311","repositories_listed":0,"syntology":null},{"url":null,"slug":"cs-dialogue-a-104-hour-dataset-of-spontaneous","title":"CS-Dialogue: A 104-Hour Dataset of Spontaneous Mandarin-English Code-Switching Dialogues for Speech Recognition","date":"2025-02-26","arxiv_id":"2502.18913","repositories_listed":0,"syntology":null},{"url":null,"slug":"nexus-o-an-omni-perceptive-and-interactive","title":"Nexus: An Omni-Perceptive And -Interactive Model for Language, Audio, And Vision","date":"2025-02-26","arxiv_id":"2503.01879","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-gender-disparities-in-automatic","title":"Exploring Gender Disparities in Automatic Speech Recognition Technology","date":"2025-02-25","arxiv_id":"2502.18434","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-speech-understanding-and-generation","title":"Balancing Speech Understanding and Generation Using Continual Pre-training for Codec-based Speech LLM","date":"2025-02-24","arxiv_id":"2502.16897","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-and-sparse-model-merging-for-multi","title":"Low-Rank and Sparse Model Merging for Multi-Lingual Speech Recognition and Translation","date":"2025-02-24","arxiv_id":"2502.17380","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-zero-shot-rare-word-recognition","title":"Understanding Zero-shot Rare Word Recognition Improvements Through LLM Integration","date":"2025-02-22","arxiv_id":"2502.16142","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-large-language-models-with","title":"Enhancing Speech Large Language Models with Prompt-Aware Mixture of Audio Encoders","date":"2025-02-21","arxiv_id":"2502.15178","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-speech-recognition","title":"Retrieval-Augmented Speech Recognition Approach for Domain Challenges","date":"2025-02-21","arxiv_id":"2502.15264","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-esethu-framework-reimagining-sustainable","title":"The Esethu Framework: Reimagining Sustainable Dataset Governance and Curation for Low-Resource Languages","date":"2025-02-21","arxiv_id":"2502.15916","repositories_listed":0,"syntology":null},{"url":null,"slug":"moshi-moshi-a-model-selection-hijacking","title":"Moshi Moshi? A Model Selection Hijacking Adversarial Attack","date":"2025-02-20","arxiv_id":"2502.14586","repositories_listed":0,"syntology":null},{"url":null,"slug":"wavrag-audio-integrated-retrieval-augmented","title":"WavRAG: Audio-Integrated Retrieval Augmented Generation for Spoken Dialogue Models","date":"2025-02-20","arxiv_id":"2502.14727","repositories_listed":0,"syntology":null},{"url":null,"slug":"adopting-whisper-for-confidence-estimation","title":"Adopting Whisper for Confidence Estimation","date":"2025-02-19","arxiv_id":"2502.13446","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-automatic-speech-recognition","title":"Benchmarking Automatic Speech Recognition coupled LLM Modules for Medical Diagnostics","date":"2025-02-18","arxiv_id":"2502.13982","repositories_listed":0,"syntology":null},{"url":null,"slug":"gesture-aware-zero-shot-speech-recognition","title":"Gesture-Aware Zero-Shot Speech Recognition for Patients with Language Disorders","date":"2025-02-18","arxiv_id":"2502.13983","repositories_listed":0,"syntology":null},{"url":null,"slug":"lost-in-transcription-found-in-distribution","title":"Lost in Transcription, Found in Distribution Shift: Demystifying Hallucination in Speech Foundation Models","date":"2025-02-18","arxiv_id":"2502.12414","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuro-oscillatory-models-of-cortical-speech","title":"Neuro-oscillatory models of cortical speech processing","date":"2025-02-18","arxiv_id":"2502.12935","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-robust-approximation-of-asr-metrics","title":"On the Robust Approximation of ASR Metrics","date":"2025-02-18","arxiv_id":"2502.12408","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-ft-merging-pre-trained-and-fine-tuned","title":"Speech-FT: Merging Pre-trained And Fine-Tuned Speech Representation Models For Cross-Task Generalization","date":"2025-02-18","arxiv_id":"2502.12672","repositories_listed":0,"syntology":null},{"url":null,"slug":"naturall2s-end-to-end-high-quality","title":"NaturalL2S: End-to-End High-quality Multispeaker Lip-to-Speech Synthesis with Differential Digital Signal Processing","date":"2025-02-17","arxiv_id":"2502.12002","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-exploration-with-gpt-4o-voice","title":"A Preliminary Exploration with GPT-4o Voice Mode","date":"2025-02-14","arxiv_id":"2502.09940","repositories_listed":0,"syntology":null},{"url":null,"slug":"microphone-array-geometry-independent-multi","title":"Microphone Array Geometry Independent Multi-Talker Distant ASR: NTT System for the DASR Task of the CHiME-8 Challenge","date":"2025-02-14","arxiv_id":"2502.09859","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtlm-an-innovative-language-model-training","title":"MTLM: Incorporating Bidirectional Text Information to Enhance Language Model Training in Speech Recognition Systems","date":"2025-02-14","arxiv_id":"2502.10058","repositories_listed":0,"syntology":null},{"url":null,"slug":"owls-scaling-laws-for-multilingual-speech","title":"OWLS: Scaling Laws for Multilingual Speech Recognition and Translation Models","date":"2025-02-14","arxiv_id":"2502.10373","repositories_listed":0,"syntology":null},{"url":null,"slug":"shortcut-learning-susceptibility-in-vision","title":"Shortcut Learning Susceptibility in Vision Classifiers","date":"2025-02-13","arxiv_id":"2502.09150","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-analysis-of-asr-errors-for-children","title":"Causal Analysis of ASR Errors for Children: Quantifying the Impact of Physiological, Cognitive, and Extrinsic Factors","date":"2025-02-12","arxiv_id":"2502.08587","repositories_listed":0,"syntology":null},{"url":null,"slug":"mohave-mixture-of-hierarchical-audio-visual","title":"MoHAVE: Mixture of Hierarchical Audio-Visual Experts for Robust Speech Recognition","date":"2025-02-11","arxiv_id":"2502.10447","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-speech-translation-with","title":"Speech to Speech Translation with Translatotron: A State of the Art Review","date":"2025-02-09","arxiv_id":"2502.05980","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-standard-and-dialectal-frisian-asr","title":"Evaluating Standard and Dialectal Frisian ASR: Multilingual Fine-tuning and Language Identification for Improved Low-resource Performance","date":"2025-02-07","arxiv_id":"2502.04883","repositories_listed":0,"syntology":null},{"url":null,"slug":"koel-tts-enhancing-llm-based-speech","title":"Koel-TTS: Enhancing LLM based Speech Generation with Preference Alignment and Classifier Free Guidance","date":"2025-02-07","arxiv_id":"2502.05236","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-operations-for-visual-speech","title":"Lightweight Operations for Visual Speech Recognition","date":"2025-02-07","arxiv_id":"2502.04834","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrispeech-dialog-a-benchmark-dataset-for","title":"Afrispeech-Dialog: A Benchmark Dataset for Spontaneous English Conversations in Healthcare and Beyond","date":"2025-02-06","arxiv_id":"2502.03945","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligner-encoders-self-attention-transformers","title":"Aligner-Encoders: Self-Attention Transformers Can Be Self-Transducers","date":"2025-02-06","arxiv_id":"2502.05232","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-differentiable-alignment-framework-for","title":"A Differentiable Alignment Framework for Sequence-to-Sequence Modeling via Optimal Transport","date":"2025-02-03","arxiv_id":"2502.01588","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapter-based-multi-agent-avsr-extension-for","title":"Adapter-Based Multi-Agent AVSR Extension for Pre-Trained ASR Models","date":"2025-02-03","arxiv_id":"2502.01709","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-dro-robust-optimization-for-reducing","title":"CTC-DRO: Robust Optimization for Reducing Language Disparities in Speech Recognition","date":"2025-02-03","arxiv_id":"2502.01777","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-norm-based-fine-tuning-for-backdoor","title":"Gradient Norm-based Fine-Tuning for Backdoor Defense in Automatic Speech Recognition","date":"2025-02-03","arxiv_id":"2502.01152","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-mispronunciation-pattern","title":"Data-Driven Mispronunciation Pattern Discovery for Robust Speech Recognition","date":"2025-02-01","arxiv_id":"2502.00583","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-end-to-end-is-overkill-rethinking","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","date":"2025-02-01","arxiv_id":"2502.00377","repositories_listed":0,"syntology":null},{"url":"/paper/dypcl-dynamic-phoneme-level-contrastive","slug":"dypcl-dynamic-phoneme-level-contrastive","title":"DyPCL: Dynamic Phoneme-level Contrastive Learning for Dysarthric Speech Recognition","date":"2025-01-31","arxiv_id":"2501.19010","repositories_listed":0,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dypcl-dynamic-phoneme-level-contrastive#ran","syntology_url":"https://syntology.ai/paper/2501.19010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19010"}},"official":null}},{"url":null,"slug":"language-bias-in-self-supervised-learning-for","title":"Language Bias in Self-Supervised Learning For Automatic Speech Recognition","date":"2025-01-31","arxiv_id":"2501.19321","repositories_listed":0,"syntology":null}],"record_sha256":"f38ba5bdd7ef0e404c83e6c5390f52ca7e50f4c76a39569469a03c5c656c4c42","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}