{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/17","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":58,"rows_per_page":100,"rows":[1601,1700],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/16","next":"/task/speech-recognition-1/papers/18","papers":[{"url":null,"slug":"tiny-align-bridging-automatic-speech","title":"Tiny-Align: Bridging Automatic Speech Recognition and Large Language Model on the Edge","date":"2024-11-21","arxiv_id":"2411.13766","repositories_listed":0,"syntology":null},{"url":null,"slug":"cafe-a-novel-code-switching-dataset-for","title":"CAFE A Novel Code switching Dataset for Algerian Dialect French and English","date":"2024-11-20","arxiv_id":"2411.13424","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-statistical-methods-to-pre-trained","title":"From Statistical Methods to Pre-Trained Models; A Survey on Automatic Speech Recognition for Resource Scarce Urdu Language","date":"2024-11-20","arxiv_id":"2411.14493","repositories_listed":0,"syntology":null},{"url":null,"slug":"hard-synth-synthesizing-diverse-hard-samples","title":"Hard-Synth: Synthesizing Diverse Hard Samples for ASR using Zero-Shot TTS and LLM","date":"2024-11-20","arxiv_id":"2411.13159","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-advanced-speech-signal-processing-a","title":"Towards Advanced Speech Signal Processing: A Statistical Perspective on Convolution-Based Architectures and its Applications","date":"2024-11-20","arxiv_id":"2411.18636","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisper-finetuning-on-nepali-language","title":"Whisper Finetuning on Nepali Language","date":"2024-11-19","arxiv_id":"2411.12587","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-speech-analysis-and-correction-tool","title":"A Novel Speech Analysis and Correction Tool for Arabic-Speaking Children","date":"2024-11-18","arxiv_id":"2411.13592","repositories_listed":0,"syntology":null},{"url":null,"slug":"inter-linguistic-phonetic-composition-ipc-a","title":"Inter-linguistic Phonetic Composition (IPC): A Theoretical and Computational Approach to Enhance Second Language Pronunciation","date":"2024-11-17","arxiv_id":"2411.10927","repositories_listed":0,"syntology":null},{"url":null,"slug":"dimodif-discourse-modality-information","title":"DiMoDif: Discourse Modality-information Differentiation for Audio-visual Deepfake Detection and Localization","date":"2024-11-15","arxiv_id":"2411.10193","repositories_listed":0,"syntology":null},{"url":null,"slug":"systolic-arrays-and-structured-pruning-co","title":"Systolic Arrays and Structured Pruning Co-design for Efficient Transformers in Edge Systems","date":"2024-11-15","arxiv_id":"2411.10285","repositories_listed":0,"syntology":null},{"url":null,"slug":"everyone-deserves-their-voice-to-be-heard","title":"Everyone deserves their voice to be heard: Analyzing Predictive Gender Bias in ASR Models Applied to Dutch Speech Data","date":"2024-11-14","arxiv_id":"2411.09431","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-adversarial-attacks-against-asr","title":"Transferable Adversarial Attacks against ASR","date":"2024-11-14","arxiv_id":"2411.09220","repositories_listed":0,"syntology":null},{"url":null,"slug":"dcf-ds-deep-cascade-fusion-of-diarization-and","title":"DCF-DS: Deep Cascade Fusion of Diarization and Separation for Speech Recognition under Realistic Single-Channel Conditions","date":"2024-11-11","arxiv_id":"2411.06667","repositories_listed":0,"syntology":null},{"url":null,"slug":"multistage-fine-tuning-strategies-for","title":"Multistage Fine-tuning Strategies for Automatic Speech Recognition in Low-resource Languages","date":"2024-11-07","arxiv_id":"2411.04573","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-aac-software-for-dysarthric","title":"Enhancing AAC Software for Dysarthric Speakers in e-Health Settings: An Evaluation Using TORGO","date":"2024-11-01","arxiv_id":"2411.00980","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-contextual-speech-recognition","title":"Optimizing Contextual Speech Recognition Using Vector Quantization for Efficient Retrieval","date":"2024-11-01","arxiv_id":"2411.00664","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-is-more-than-words-do-speech-to-text","title":"Speech is More Than Words: Do Speech-to-Text Translation Systems Leverage Prosody?","date":"2024-10-31","arxiv_id":"2410.24019","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-polish-automatic-speech","title":"Augmenting Polish Automatic Speech Recognition System With Synthetic Data","date":"2024-10-30","arxiv_id":"2410.22903","repositories_listed":0,"syntology":null},{"url":null,"slug":"run-time-adaptation-of-neural-beamforming-for","title":"Run-Time Adaptation of Neural Beamforming for Robust Speech Dereverberation and Denoising","date":"2024-10-30","arxiv_id":"2410.22805","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-beamforming-and-speaker-attributed-asr","title":"Joint Beamforming and Speaker-Attributed ASR for Real Distant-Microphone Meeting Transcription","date":"2024-10-29","arxiv_id":"2410.21849","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-tool-usage-for-real-time-agents","title":"Asynchronous Tool Usage for Real-Time Agents","date":"2024-10-28","arxiv_id":"2410.21620","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-standalone-trustworthy-voice","title":"Multilingual Standalone Trustworthy Voice-Based Social Network for Disaster Situations","date":"2024-10-28","arxiv_id":"2411.08889","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-based-emotion-recognition","title":"Improving Speech-based Emotion Recognition with Contextual Utterance Analysis and LLMs","date":"2024-10-27","arxiv_id":"2410.20334","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-speech-large-language-models","title":"A Survey on Speech Large Language Models","date":"2024-10-24","arxiv_id":"2410.18908","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-biasing-to-improve-domain-specific","title":"Contextual Biasing to Improve Domain-specific Custom Vocabulary Audio Transcription without Explicit Fine-Tuning of Whisper Model","date":"2024-10-24","arxiv_id":"2410.18363","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-improving-automatic-speech","title":"Evaluating and Improving Automatic Speech Recognition Systems for Korean Meteorological Experts","date":"2024-10-24","arxiv_id":"2410.18444","repositories_listed":0,"syntology":null},{"url":null,"slug":"we-augmented-whisper-with-knn-and-you-won-t","title":"kNN For Whisper And Its Effect On Bias And Speaker Adaptation","date":"2024-10-24","arxiv_id":"2410.18850","repositories_listed":0,"syntology":null},{"url":null,"slug":"elaichi-enhancing-low-resource-tts-by","title":"ELAICHI: Enhancing Low-resource TTS by Addressing Infrequent and Low-frequency Character Bigrams","date":"2024-10-23","arxiv_id":"2410.17901","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoasr-debiasing-asrs-through-selective","title":"DENOASR: Debiasing ASRs through Selective Denoising","date":"2024-10-22","arxiv_id":"2410.16712","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-low-resource-asr-through-versatile","title":"Enhancing Low-Resource ASR through Versatile TTS: Bridging the Data Gap","date":"2024-10-22","arxiv_id":"2410.16726","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-automatic-speech-recognition-with","title":"Improving Automatic Speech Recognition with Decoder-Centric Regularisation in Encoder-Decoder Models","date":"2024-10-22","arxiv_id":"2410.17437","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-model-optimization-over-multiple","title":"Acoustic Model Optimization over Multiple Data Sources: Merging and Valuation","date":"2024-10-21","arxiv_id":"2410.15620","repositories_listed":0,"syntology":null},{"url":null,"slug":"interventional-speech-noise-injection-for-asr","title":"Interventional Speech Noise Injection for ASR Generalizable Spoken Language Understanding","date":"2024-10-21","arxiv_id":"2410.15609","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-transformer-based-automatic-speech","title":"End-to-End Transformer-based Automatic Speech Recognition for Northern Kurdish: A Pioneering Approach","date":"2024-10-19","arxiv_id":"2410.16330","repositories_listed":0,"syntology":null},{"url":null,"slug":"ac-mix-self-supervised-adaptation-for-low","title":"AC-Mix: Self-Supervised Adaptation for Low-Resource Automatic Speech Recognition using Agnostic Contrastive Mixup","date":"2024-10-18","arxiv_id":"2410.14910","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-approaches-to-arabic-english","title":"Computational Approaches to Arabic-English Code-Switching","date":"2024-10-17","arxiv_id":"2410.13318","repositories_listed":0,"syntology":null},{"url":null,"slug":"failing-forward-improving-generative-error","title":"Failing Forward: Improving Generative Error Correction for ASR with Synthetic Data and Retrieval Augmentation","date":"2024-10-17","arxiv_id":"2410.13198","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-adaptation-of","title":"Parameter-efficient Adaptation of Multilingual Multimodal Models for Low-resource ASR","date":"2024-10-17","arxiv_id":"2410.13445","repositories_listed":0,"syntology":null},{"url":null,"slug":"roadmap-towards-superhuman-speech","title":"Roadmap towards Superhuman Speech Understanding using Large Language Models","date":"2024-10-17","arxiv_id":"2410.13268","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-speaker-representation-for","title":"Investigation of Speaker Representation for Target-Speaker Speech Processing","date":"2024-10-15","arxiv_id":"2410.11243","repositories_listed":0,"syntology":null},{"url":null,"slug":"character-aware-audio-visual-subtitling-in","title":"Character-aware audio-visual subtitling in context","date":"2024-10-14","arxiv_id":"2410.11068","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-materia-speech-recognition","title":"In-Materia Speech Recognition","date":"2024-10-14","arxiv_id":"2410.10434","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-of-nlp-in-kenya-a-survey","title":"State of NLP in Kenya: A Survey","date":"2024-10-13","arxiv_id":"2410.09948","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-with-bert-and","title":"Automatic Speech Recognition with BERT and CTC Transformers: A Review","date":"2024-10-12","arxiv_id":"2410.09456","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-indonesian-automatic-speech","title":"Enhancing Indonesian Automatic Speech Recognition: Evaluating Multilingual Models with Diverse Speech Variabilities","date":"2024-10-11","arxiv_id":"2410.08828","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniglyph-a-seven-segment-script-for-universal","title":"UniGlyph: A Seven-Segment Script for Universal Language Representation","date":"2024-10-11","arxiv_id":"2410.08974","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-rank-no-more-low-rank-weight-training","title":"Full-Rank No More: Low-Rank Weight Training for Modern Speech Recognition Models","date":"2024-10-10","arxiv_id":"2410.07771","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-stage-transliteration-approach-to","title":"A two-stage transliteration approach to improve performance of a multilingual ASR","date":"2024-10-09","arxiv_id":"2410.14709","repositories_listed":0,"syntology":null},{"url":null,"slug":"advocating-character-error-rate-for","title":"Advocating Character Error Rate for Multilingual ASR Evaluation","date":"2024-10-09","arxiv_id":"2410.07400","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ustc-nercslip-systems-for-the-chime-8-1","title":"The USTC-NERCSLIP Systems for the CHiME-8 MMCSG Challenge","date":"2024-10-08","arxiv_id":"2410.05986","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-screening-for-children-with-speech","title":"Automatic Screening for Children with Speech Disorder using Automatic Speech Recognition: Opportunities and Challenges","date":"2024-10-07","arxiv_id":"2410.11865","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-talker-identity-aids-with","title":"Incorporating Talker Identity Aids With Improving Speech Recognition in Adversarial Environments","date":"2024-10-07","arxiv_id":"2410.05423","repositories_listed":0,"syntology":null},{"url":null,"slug":"casablanca-data-and-models-for-multidialectal","title":"Casablanca: Data and Models for Multidialectal Arabic Speech Recognition","date":"2024-10-06","arxiv_id":"2410.04527","repositories_listed":0,"syntology":null},{"url":null,"slug":"punctuation-prediction-for-polish-texts-using","title":"Punctuation Prediction for Polish Texts using Transformers","date":"2024-10-06","arxiv_id":"2410.04621","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancement-of-dysarthric-speech","title":"Enhancement of Dysarthric Speech Reconstruction by Contrastive Learning","date":"2024-10-05","arxiv_id":"2410.04092","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ocon-model-an-old-but-green-solution-for","title":"The OCON model: an old but green solution for distributable supervised classification for acoustic monitoring in smart cities","date":"2024-10-05","arxiv_id":"2410.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"reverb-open-source-asr-and-diarization-from","title":"Reverb: Open-Source ASR and Diarization from Rev","date":"2024-10-04","arxiv_id":"2410.03930","repositories_listed":0,"syntology":null},{"url":null,"slug":"team-mts-automin-2021-an-overview-of-existing","title":"Team MTS @ AutoMin 2021: An Overview of Existing Summarization Approaches and Comparison to Unsupervised Summarization Techniques","date":"2024-10-04","arxiv_id":"2410.03412","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pilot-study-of-applying-sequence-to","title":"A Pilot Study of Applying Sequence-to-Sequence Voice Conversion to Evaluate the Intelligibility of L2 Speech Using a Native Speaker's Shadowings","date":"2024-10-03","arxiv_id":"2410.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithms-for-automatic-accentuation-and","title":"Algorithms For Automatic Accentuation And Transcription Of Russian Texts In Speech Recognition Systems","date":"2024-10-03","arxiv_id":"2410.02538","repositories_listed":0,"syntology":null},{"url":null,"slug":"convolutional-variational-autoencoders-for-1","title":"Convolutional Variational Autoencoders for Spectrogram Compression in Automatic Speech Recognition","date":"2024-10-03","arxiv_id":"2410.02560","repositories_listed":0,"syntology":null},{"url":null,"slug":"three-in-one-fast-and-accurate-transducer-for","title":"HAINAN: Fast and Accurate Transducer for Hybrid-Autoregressive ASR","date":"2024-10-03","arxiv_id":"2410.02597","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-streaming-llm-for-speech","title":"Efficient Streaming LLM for Speech Recognition","date":"2024-10-02","arxiv_id":"2410.03752","repositories_listed":0,"syntology":null},{"url":null,"slug":"spoken-grammar-assessment-using-llm","title":"Spoken Grammar Assessment Using LLM","date":"2024-10-02","arxiv_id":"2410.01579","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-the-ika","title":"Automatic Speech Recognition for the Ika Language","date":"2024-10-01","arxiv_id":"2410.00940","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-free-training-for-transducer-based","title":"Alignment-Free Training for Transducer-based Multi-Talker ASR","date":"2024-09-30","arxiv_id":"2409.20301","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-hybrid-autoregressive-transducer","title":"Boosting Hybrid Autoregressive Transducer-based ASR with Internal Acoustic Model Training and Dual Blank Thresholding","date":"2024-09-30","arxiv_id":"2409.20313","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-speech-recognition-and-end-of","title":"Predictive Speech Recognition and End-of-Utterance Detection Towards Spoken Dialog Systems","date":"2024-09-30","arxiv_id":"2409.19990","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-long-form-speech-recognition-for","title":"Efficient Long-Form Speech Recognition for General Speech In-Context Learning","date":"2024-09-29","arxiv_id":"2409.19757","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-automatic-speech-recognition-for","title":"Fine-Tuning Automatic Speech Recognition for People with Parkinson's: An Effective Strategy for Enhancing Speech Technology Accessibility","date":"2024-09-29","arxiv_id":"2409.19818","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-analysis-of-audio-visual-tasks","title":"Quantitative Analysis of Audio-Visual Tasks: An Information-Theoretic Perspective","date":"2024-09-29","arxiv_id":"2409.19575","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-clustering-techniques-for-speech","title":"Advanced Clustering Techniques for Speech Signal Enhancement: A Review and Metanalysis of Fuzzy C-Means, K-Means, and Kernel Fuzzy C-Means Methods","date":"2024-09-28","arxiv_id":"2409.19448","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gen-ai-framework-for-medical-note","title":"A GEN AI Framework for Medical Note Generation","date":"2024-09-27","arxiv_id":"2410.01841","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-mamba-long-context-speech-recognition","title":"Speech-Mamba: Long-Context Speech Recognition with Selective State Spaces Models","date":"2024-09-27","arxiv_id":"2409.18654","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-transformers-in-pre-trained-lm-a-good-asr","title":"Are Transformers in Pre-trained LM A Good ASR Encoder? An Empirical Study","date":"2024-09-26","arxiv_id":"2409.17750","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-clas-deep-contextual-listen-attend-and","title":"Deep CLAS: Deep Contextual Listen, Attend and Spell","date":"2024-09-26","arxiv_id":"2409.17603","repositories_listed":0,"syntology":null},{"url":null,"slug":"paraformer-v2-an-improved-non-autoregressive","title":"Paraformer-v2: An improved non-autoregressive transformer for noise-robust speech recognition","date":"2024-09-26","arxiv_id":"2409.17746","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-the-role-of-pretraining-in-direct","title":"Unveiling the Role of Pretraining in Direct Speech Translation","date":"2024-09-26","arxiv_id":"2409.18044","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-connect-speech-foundation-models-and","title":"How to Connect Speech Foundation Models and Large Language Models? What Matters and What Does Not","date":"2024-09-25","arxiv_id":"2409.17044","repositories_listed":0,"syntology":null},{"url":null,"slug":"mt2kd-towards-a-general-purpose-encoder-for","title":"MT2KD: Towards A General-Purpose Encoder for Speech, Speaker, and Audio Events","date":"2024-09-25","arxiv_id":"2409.17010","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-rescoring-with-large","title":"Speech Recognition Rescoring with Large Speech-Text Foundation Models","date":"2024-09-25","arxiv_id":"2409.16654","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-code-switching-asr-with-mixture-of","title":"Boosting Code-Switching ASR with Mixture of Experts Enhanced Speech-Conditioned LLM","date":"2024-09-24","arxiv_id":"2409.15905","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-speech-and-text-enhancing-asr-with","title":"Bridging Speech and Text: Enhancing ASR with Pinyin-to-Character Pre-training in LLMs","date":"2024-09-24","arxiv_id":"2409.16005","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypothesis-clustering-and-merging-novel","title":"Hypothesis Clustering and Merging: Novel MultiTalker Speech Recognition with Speaker Tokens","date":"2024-09-24","arxiv_id":"2409.15732","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-acoustic-features-for-robust-asr","title":"Revisiting Acoustic Features for Robust ASR","date":"2024-09-24","arxiv_id":"2409.16399","repositories_listed":0,"syntology":null},{"url":null,"slug":"spelling-correction-through-rewriting-of-non","title":"Spelling Correction through Rewriting of Non-Autoregressive ASR Lattices","date":"2024-09-24","arxiv_id":"2409.16469","repositories_listed":0,"syntology":null},{"url":null,"slug":"strong-alone-stronger-together-synergizing","title":"Strong Alone, Stronger Together: Synergizing Modality-Binding Foundation Models with Optimal Transport for Non-Verbal Emotion Recognition","date":"2024-09-21","arxiv_id":"2409.14221","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-dense-retrieval-approach-for","title":"A Multimodal Dense Retrieval Approach for Speech-Based Open-Domain Question Answering","date":"2024-09-20","arxiv_id":"2409.13483","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-streaming-transducer-asr-prototyping-via","title":"Fast Streaming Transducer ASR Prototyping via Knowledge Distillation with Whisper","date":"2024-09-20","arxiv_id":"2409.13499","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-should-understand-pinyin","title":"Large Language Model Should Understand Pinyin for Chinese ASR Error Correction","date":"2024-09-20","arxiv_id":"2409.13262","repositories_listed":0,"syntology":null},{"url":null,"slug":"lm-assisted-keyword-biasing-with-aho-corasick","title":"LM-assisted keyword biasing with Aho-Corasick algorithm for Transducer-based ASR","date":"2024-09-20","arxiv_id":"2409.13514","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-and-tokens-benchmarking-end-to-end","title":"Time and Tokens: Benchmarking End-to-End Speech Dysfluency Detection","date":"2024-09-20","arxiv_id":"2409.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-synthetic-training-data-for-speech","title":"Enhancing Synthetic Training Data for Speech Commands: From ASR-Based Filtering to Domain Adaptation in SSL Latent Space","date":"2024-09-19","arxiv_id":"2409.12745","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-speech-recognition-for-children","title":"Personalized Speech Recognition for Children with Test-Time Adaptation","date":"2024-09-19","arxiv_id":"2409.13095","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-audiovisual-speech-recognition-models","title":"Robust Audiovisual Speech Recognition Models with Mixture-of-Experts","date":"2024-09-19","arxiv_id":"2409.12370","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-cat-speaker-informed-speech-embeddings","title":"META-CAT: Speaker-Informed Speech Embeddings via Meta Information Concatenation for Multi-talker ASR","date":"2024-09-18","arxiv_id":"2409.12352","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-joint-spectro-temporal-relational-thinking","title":"A Joint Spectro-Temporal Relational Thinking Based Acoustic Modeling Framework","date":"2024-09-17","arxiv_id":"2409.15357","repositories_listed":0,"syntology":null},{"url":null,"slug":"bio-inspired-mamba-temporal-locality-and","title":"Bio-Inspired Mamba: Temporal Locality and Bioplausible Learning in Selective State Space Models","date":"2024-09-17","arxiv_id":"2409.11263","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-thought-prompting-for-speech","title":"Chain-of-Thought Prompting for Speech Translation","date":"2024-09-17","arxiv_id":"2409.11538","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-low-resource-language-and","title":"Enhancing Low-Resource Language and Instruction Following Capabilities of Audio Language Models","date":"2024-09-17","arxiv_id":"2409.10999","repositories_listed":0,"syntology":null}],"record_sha256":"d521d9f4cf1b262bdf02253a3df24bbe3b581da70f6ffc1cfcba0cbdf42b93ef","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}