{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/9","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":31,"rows_per_page":100,"rows":[801,900],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/8","next":"/task/automatic-speech-recognition/papers/10","papers":[{"url":null,"slug":"denoasr-debiasing-asrs-through-selective","title":"DENOASR: Debiasing ASRs through Selective Denoising","date":"2024-10-22","arxiv_id":"2410.16712","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-low-resource-asr-through-versatile","title":"Enhancing Low-Resource ASR through Versatile TTS: Bridging the Data Gap","date":"2024-10-22","arxiv_id":"2410.16726","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-automatic-speech-recognition-with","title":"Improving Automatic Speech Recognition with Decoder-Centric Regularisation in Encoder-Decoder Models","date":"2024-10-22","arxiv_id":"2410.17437","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-model-optimization-over-multiple","title":"Acoustic Model Optimization over Multiple Data Sources: Merging and Valuation","date":"2024-10-21","arxiv_id":"2410.15620","repositories_listed":0,"syntology":null},{"url":null,"slug":"interventional-speech-noise-injection-for-asr","title":"Interventional Speech Noise Injection for ASR Generalizable Spoken Language Understanding","date":"2024-10-21","arxiv_id":"2410.15609","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-transformer-based-automatic-speech","title":"End-to-End Transformer-based Automatic Speech Recognition for Northern Kurdish: A Pioneering Approach","date":"2024-10-19","arxiv_id":"2410.16330","repositories_listed":0,"syntology":null},{"url":null,"slug":"ac-mix-self-supervised-adaptation-for-low","title":"AC-Mix: Self-Supervised Adaptation for Low-Resource Automatic Speech Recognition using Agnostic Contrastive Mixup","date":"2024-10-18","arxiv_id":"2410.14910","repositories_listed":0,"syntology":null},{"url":null,"slug":"failing-forward-improving-generative-error","title":"Failing Forward: Improving Generative Error Correction for ASR with Synthetic Data and Retrieval Augmentation","date":"2024-10-17","arxiv_id":"2410.13198","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-adaptation-of","title":"Parameter-efficient Adaptation of Multilingual Multimodal Models for Low-resource ASR","date":"2024-10-17","arxiv_id":"2410.13445","repositories_listed":0,"syntology":null},{"url":null,"slug":"roadmap-towards-superhuman-speech","title":"Roadmap towards Superhuman Speech Understanding using Large Language Models","date":"2024-10-17","arxiv_id":"2410.13268","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-with-bert-and","title":"Automatic Speech Recognition with BERT and CTC Transformers: A Review","date":"2024-10-12","arxiv_id":"2410.09456","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-indonesian-automatic-speech","title":"Enhancing Indonesian Automatic Speech Recognition: Evaluating Multilingual Models with Diverse Speech Variabilities","date":"2024-10-11","arxiv_id":"2410.08828","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-stage-transliteration-approach-to","title":"A two-stage transliteration approach to improve performance of a multilingual ASR","date":"2024-10-09","arxiv_id":"2410.14709","repositories_listed":0,"syntology":null},{"url":null,"slug":"advocating-character-error-rate-for","title":"Advocating Character Error Rate for Multilingual ASR Evaluation","date":"2024-10-09","arxiv_id":"2410.07400","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-screening-for-children-with-speech","title":"Automatic Screening for Children with Speech Disorder using Automatic Speech Recognition: Opportunities and Challenges","date":"2024-10-07","arxiv_id":"2410.11865","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ocon-model-an-old-but-green-solution-for","title":"The OCON model: an old but green solution for distributable supervised classification for acoustic monitoring in smart cities","date":"2024-10-05","arxiv_id":"2410.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithms-for-automatic-accentuation-and","title":"Algorithms For Automatic Accentuation And Transcription Of Russian Texts In Speech Recognition Systems","date":"2024-10-03","arxiv_id":"2410.02538","repositories_listed":0,"syntology":null},{"url":null,"slug":"convolutional-variational-autoencoders-for-1","title":"Convolutional Variational Autoencoders for Spectrogram Compression in Automatic Speech Recognition","date":"2024-10-03","arxiv_id":"2410.02560","repositories_listed":0,"syntology":null},{"url":null,"slug":"spoken-grammar-assessment-using-llm","title":"Spoken Grammar Assessment Using LLM","date":"2024-10-02","arxiv_id":"2410.01579","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-the-ika","title":"Automatic Speech Recognition for the Ika Language","date":"2024-10-01","arxiv_id":"2410.00940","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-free-training-for-transducer-based","title":"Alignment-Free Training for Transducer-based Multi-Talker ASR","date":"2024-09-30","arxiv_id":"2409.20301","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-speech-recognition-and-end-of","title":"Predictive Speech Recognition and End-of-Utterance Detection Towards Spoken Dialog Systems","date":"2024-09-30","arxiv_id":"2409.19990","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-long-form-speech-recognition-for","title":"Efficient Long-Form Speech Recognition for General Speech In-Context Learning","date":"2024-09-29","arxiv_id":"2409.19757","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-automatic-speech-recognition-for","title":"Fine-Tuning Automatic Speech Recognition for People with Parkinson's: An Effective Strategy for Enhancing Speech Technology Accessibility","date":"2024-09-29","arxiv_id":"2409.19818","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-gen-ai-framework-for-medical-note","title":"A GEN AI Framework for Medical Note Generation","date":"2024-09-27","arxiv_id":"2410.01841","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-transformers-in-pre-trained-lm-a-good-asr","title":"Are Transformers in Pre-trained LM A Good ASR Encoder? An Empirical Study","date":"2024-09-26","arxiv_id":"2409.17750","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-clas-deep-contextual-listen-attend-and","title":"Deep CLAS: Deep Contextual Listen, Attend and Spell","date":"2024-09-26","arxiv_id":"2409.17603","repositories_listed":0,"syntology":null},{"url":null,"slug":"mt2kd-towards-a-general-purpose-encoder-for","title":"MT2KD: Towards A General-Purpose Encoder for Speech, Speaker, and Audio Events","date":"2024-09-25","arxiv_id":"2409.17010","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-rescoring-with-large","title":"Speech Recognition Rescoring with Large Speech-Text Foundation Models","date":"2024-09-25","arxiv_id":"2409.16654","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-code-switching-asr-with-mixture-of","title":"Boosting Code-Switching ASR with Mixture of Experts Enhanced Speech-Conditioned LLM","date":"2024-09-24","arxiv_id":"2409.15905","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-speech-and-text-enhancing-asr-with","title":"Bridging Speech and Text: Enhancing ASR with Pinyin-to-Character Pre-training in LLMs","date":"2024-09-24","arxiv_id":"2409.16005","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-acoustic-features-for-robust-asr","title":"Revisiting Acoustic Features for Robust ASR","date":"2024-09-24","arxiv_id":"2409.16399","repositories_listed":0,"syntology":null},{"url":null,"slug":"spelling-correction-through-rewriting-of-non","title":"Spelling Correction through Rewriting of Non-Autoregressive ASR Lattices","date":"2024-09-24","arxiv_id":"2409.16469","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-dense-retrieval-approach-for","title":"A Multimodal Dense Retrieval Approach for Speech-Based Open-Domain Question Answering","date":"2024-09-20","arxiv_id":"2409.13483","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-streaming-transducer-asr-prototyping-via","title":"Fast Streaming Transducer ASR Prototyping via Knowledge Distillation with Whisper","date":"2024-09-20","arxiv_id":"2409.13499","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-and-tokens-benchmarking-end-to-end","title":"Time and Tokens: Benchmarking End-to-End Speech Dysfluency Detection","date":"2024-09-20","arxiv_id":"2409.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-speech-recognition-for-children","title":"Personalized Speech Recognition for Children with Test-Time Adaptation","date":"2024-09-19","arxiv_id":"2409.13095","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-cat-speaker-informed-speech-embeddings","title":"META-CAT: Speaker-Informed Speech Embeddings via Meta Information Concatenation for Multi-talker ASR","date":"2024-09-18","arxiv_id":"2409.12352","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-thought-prompting-for-speech","title":"Chain-of-Thought Prompting for Speech Translation","date":"2024-09-17","arxiv_id":"2409.11538","repositories_listed":0,"syntology":null},{"url":null,"slug":"ideal-llm-integrating-dual-encoders-and","title":"Ideal-LLM: Integrating Dual Encoders and Language-Adapted LLM for Multilingual Speech-to-Text","date":"2024-09-17","arxiv_id":"2409.11214","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-best-rq-a-multi-channel-speech-foundation","title":"M-BEST-RQ: A Multi-Channel Speech Foundation Model for Smart Glasses","date":"2024-09-17","arxiv_id":"2409.11494","repositories_listed":0,"syntology":null},{"url":null,"slug":"wer-we-stand-benchmarking-urdu-asr-models","title":"WER We Stand: Benchmarking Urdu ASR Models","date":"2024-09-17","arxiv_id":"2409.11252","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-augmentation-for","title":"Zero Shot Text to Speech Augmentation for Automatic Speech Recognition on Low-Resource Accented Speech Corpora","date":"2024-09-17","arxiv_id":"2409.11107","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-self-learning-framework-for","title":"An Efficient Self-Learning Framework For Interactive Spoken Dialog Systems","date":"2024-09-16","arxiv_id":"2409.10515","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-automatic-speech-recognition","title":"Augmenting Automatic Speech Recognition Models with Disfluency Detection","date":"2024-09-16","arxiv_id":"2409.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-whisper-speech-based-meta-icl-for-asr-on","title":"SMILE: Speech Meta In-Context Learning for Low-Resource Language Automatic Speech Recognition","date":"2024-09-16","arxiv_id":"2409.10429","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-based-generative-error","title":"Large Language Model Based Generative Error Correction: A Challenge and Baselines for Speech Recognition, Speaker Tagging, and Emotion Recognition","date":"2024-09-15","arxiv_id":"2409.09785","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-error-correction-using-large-language","title":"ASR Error Correction using Large Language Models","date":"2024-09-14","arxiv_id":"2409.09554","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpt-boosted-wav2vec2-0-towards-noise-robust","title":"CPT-Boosted Wav2vec2.0: Towards Noise Robust Speech Recognition for Classroom Environments","date":"2024-09-13","arxiv_id":"2409.14494","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-ssl-discrete-tokens-for","title":"Exploring SSL Discrete Tokens for Multilingual ASR","date":"2024-09-13","arxiv_id":"2409.08805","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-impact-of-data-quantity-on-asr","title":"Exploring the Impact of Data Quantity on ASR in Extremely Low-resource Languages","date":"2024-09-13","arxiv_id":"2409.08872","repositories_listed":0,"syntology":null},{"url":null,"slug":"la-rag-enhancing-llm-based-asr-accuracy-with","title":"LA-RAG:Enhancing LLM-based ASR Accuracy with Retrieval-Augmented Generation","date":"2024-09-13","arxiv_id":"2409.08597","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnings-from-curating-a-trustworthy-well","title":"Learnings from curating a trustworthy, well-annotated, and useful dataset of disordered English speech","date":"2024-09-13","arxiv_id":"2409.09190","repositories_listed":0,"syntology":null},{"url":null,"slug":"nest-rq-next-token-prediction-for-speech-self","title":"NEST-RQ: Next Token Prediction for Speech Self-Supervised Pre-Training","date":"2024-09-13","arxiv_id":"2409.08680","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-text-error-correction-for-chinese-speech","title":"Full-text Error Correction for Chinese Speech Recognition with Large Language Model","date":"2024-09-12","arxiv_id":"2409.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-ctc-based-visual-speech-recognition","title":"Enhancing CTC-Based Visual Speech Recognition","date":"2024-09-11","arxiv_id":"2409.07210","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-context-balanced-adaptation","title":"An Effective Context-Balanced Adaptation Approach for Long-Tailed Speech Recognition","date":"2024-09-10","arxiv_id":"2409.06468","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-aware-asr-error-augmentation-for","title":"Keyword-Aware ASR Error Augmentation for Robust Dialogue State Tracking","date":"2024-09-10","arxiv_id":"2409.06263","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-modularity-for-noise","title":"An investigation of modularity for noise robustness in conformer-based ASR","date":"2024-09-09","arxiv_id":"2409.05589","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-real-time-transcriptions-using","title":"Evaluation of real-time transcriptions using end-to-end ASR models","date":"2024-09-09","arxiv_id":"2409.05674","repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-2024-mandarin-stuttering","title":"Findings of the 2024 Mandarin Stuttering Event Detection and Automatic Speech Recognition Challenge","date":"2024-09-09","arxiv_id":"2409.05430","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-correction-of-named","title":"Retrieval Augmented Correction of Named Entity Speech Recognition Errors","date":"2024-09-09","arxiv_id":"2409.06062","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-wavlm-back-ends-for-speech-spoofing","title":"Exploring WavLM Back-ends for Speech Spoofing and Deepfake Detection","date":"2024-09-08","arxiv_id":"2409.05032","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-self-attention-in-self-supervised","title":"Probing self-attention in self-supervised speech models for cross-linguistic differences","date":"2024-09-04","arxiv_id":"2409.03115","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantification-of-stylistic-differences-in","title":"Quantification of stylistic differences in human- and ASR-produced transcripts of African American English","date":"2024-09-04","arxiv_id":"2409.03059","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-lost-in-normalization-exploring","title":"What is lost in Normalization? Exploring Pitfalls in Multilingual ASR Model Evaluations","date":"2024-09-04","arxiv_id":"2409.02449","repositories_listed":0,"syntology":null},{"url":null,"slug":"reassessing-noise-augmentation-methods-in-the","title":"Reassessing Noise Augmentation Methods in the Context of Adversarial Speech","date":"2024-09-03","arxiv_id":"2409.01813","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-order-preserved-optimal-transport","title":"Temporal Order Preserved Optimal Transport-based Cross-modal Knowledge Transfer Learning for ASR","date":"2024-09-03","arxiv_id":"2409.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxhakka-a-dialectally-diverse-multi-speaker","title":"VoxHakka: A Dialectally Diverse Multi-speaker Text-to-Speech System for Taiwanese Hakka","date":"2024-09-03","arxiv_id":"2409.01548","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-adaptation-of-speech","title":"Resource-Efficient Adaptation of Speech Foundation Models for Multi-Speaker ASR","date":"2024-09-02","arxiv_id":"2409.01438","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-discrete-and-continuous-space-llms","title":"Comparing Discrete and Continuous Space LLMs for Speech Recognition","date":"2024-09-01","arxiv_id":"2409.00800","repositories_listed":0,"syntology":null},{"url":null,"slug":"serialized-speech-information-guidance-with","title":"Serialized Speech Information Guidance with Overlapped Encoding Separation for Multi-Speaker Automatic Speech Recognition","date":"2024-09-01","arxiv_id":"2409.00815","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-multi-talker-asr-performance-with","title":"Advancing Multi-talker ASR Performance with Large Language Models","date":"2024-08-30","arxiv_id":"2408.17431","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-tagging-correction-with-non","title":"Speaker Tagging Correction With Non-Autoregressive Language Models","date":"2024-08-30","arxiv_id":"2409.00151","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-japanese-speech-recognition-on","title":"Benchmarking Japanese Speech Recognition on ASR-LLM Setups with Multi-Pass Augmented Generative Error Correction","date":"2024-08-29","arxiv_id":"2408.16180","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-recognition-and-detection-of","title":"Automatic recognition and detection of aphasic natural speech","date":"2024-08-26","arxiv_id":"2408.14082","repositories_listed":0,"syntology":null},{"url":null,"slug":"medsage-enhancing-robustness-of-medical","title":"MEDSAGE: Enhancing Robustness of Medical Dialogue Summarization to ASR Errors with LLM-generated Synthetic Dialogues","date":"2024-08-26","arxiv_id":"2408.14418","repositories_listed":0,"syntology":null},{"url":null,"slug":"focused-discriminative-training-for-streaming","title":"Focused Discriminative Training For Streaming CTC-Trained Automatic Speech Recognition Models","date":"2024-08-23","arxiv_id":"2408.13008","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-state-of-commercial-automatic-french","title":"The State of Commercial Automatic French Legal Speech Recognition Systems and their Impact on Court Reporters et al","date":"2024-08-21","arxiv_id":"2408.11940","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-transfer-learning-under","title":"Parameter-Efficient Transfer Learning under Federated Learning for Automatic Speech Recognition","date":"2024-08-19","arxiv_id":"2408.11873","repositories_listed":0,"syntology":null},{"url":null,"slug":"recording-for-eyes-not-echoing-to-ears","title":"Recording for Eyes, Not Echoing to Ears: Contextualized Spoken-to-Written Conversion of ASR Transcripts","date":"2024-08-19","arxiv_id":"2408.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-large-language-model-based-speech","title":"Enhancing Large Language Model-based Speech Recognition by Contextualization for Rare and Ambiguous Words","date":"2024-08-15","arxiv_id":"2408.08027","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-talker-finetuning-audio-language-model","title":"Style-Talker: Finetuning Audio Language Model and Style-Based Text-to-Speech Model for Fast Spoken Dialogue Generation","date":"2024-08-13","arxiv_id":"2408.11849","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-enhancement-for-computer-audition-an","title":"Audio Enhancement for Computer Audition -- An Iterative Training Paradigm Using Sample Importance","date":"2024-08-12","arxiv_id":"2408.06264","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-dialogue-speech-recognition-with","title":"Enhancing Dialogue Speech Recognition with Robust Contextual Awareness via Noise Representation Learning","date":"2024-08-12","arxiv_id":"2408.06043","repositories_listed":0,"syntology":null},{"url":null,"slug":"vq-ctap-cross-modal-fine-grained-sequence","title":"VQ-CTAP: Cross-Modal Fine-Grained Sequence Representation Learning for Speech Processing","date":"2024-08-11","arxiv_id":"2408.05758","repositories_listed":0,"syntology":null},{"url":"/paper/mathbridge-a-large-scale-dataset-for","slug":"mathbridge-a-large-scale-dataset-for","title":"MathBridge: A Large Corpus Dataset for Translating Spoken Mathematical Expressions into $LaTeX$ Formulas for Improved Readability","date":"2024-08-07","arxiv_id":"2408.07081","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02945","title":"Self-Supervised Learning for Multi-Channel Neural Transducer","date":"2024-08-06","arxiv_id":"2408.02945","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02978","title":"ASR-enhanced Multimodal Representation Learning for Cross-Domain Product Retrieval","date":"2024-08-06","arxiv_id":"2408.02978","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamvoice-evolving-into-end-to-end","title":"StreamVoice+: Evolving into End-to-end Streaming Zero-shot Voice Conversion","date":"2024-08-05","arxiv_id":"2408.02178","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00205","title":"Sentence-wise Speech Summarization: Task, Datasets, and End-to-End Modeling with LM Knowledge Distillation","date":"2024-08-01","arxiv_id":"2408.00205","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21414","title":"Towards interfacing large language models with ASR systems using confidence measures and prompting","date":"2024-07-31","arxiv_id":"2407.21414","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21476","title":"On the Problem of Text-To-Speech Model Selection for Synthetic Data Generation in Automatic Speech Recognition","date":"2024-07-31","arxiv_id":"2407.21476","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effect-of-purely-synthetic-training","title":"On the Effect of Purely Synthetic Training Data for Different Automatic Speech Recognition Architectures","date":"2024-07-25","arxiv_id":"2407.17997","repositories_listed":0,"syntology":null},{"url":null,"slug":"reexamining-racial-disparities-in-automatic","title":"Reexamining Racial Disparities in Automatic Speech Recognition Performance: The Role of Confounding by Provenance","date":"2024-07-19","arxiv_id":"2407.13982","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00004","title":"Handling Numeric Expressions in Automatic Speech Recognition","date":"2024-07-18","arxiv_id":"2408.00004","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-light-weight-and-efficient-punctuation-and","title":"A light-weight and efficient punctuation and word casing prediction model for on-device streaming ASR","date":"2024-07-18","arxiv_id":"2407.13142","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resourced-speech-recognition-for-iu-mien","title":"Low-Resourced Speech Recognition for Iu Mien Language via Weakly-Supervised Phoneme-based Multilingual Pre-training","date":"2024-07-18","arxiv_id":"2407.13292","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-asr-error-correction-with-conservative","title":"Robust ASR Error Correction with Conservative Data Filtering","date":"2024-07-18","arxiv_id":"2407.13300","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-binary-multiclass-paraphasia-detection","title":"Beyond Binary: Multiclass Paraphasia Detection with Generative Pretrained Transformers and End-to-End Models","date":"2024-07-16","arxiv_id":"2407.11345","repositories_listed":0,"syntology":null}],"record_sha256":"e9d507f7b12484405b88ac35cc99e1cc1cba89a943a1119f902acaf9e9a726a1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}