{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/18","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":58,"rows_per_page":100,"rows":[1701,1800],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/17","next":"/task/speech-recognition-1/papers/19","papers":[{"url":null,"slug":"ideal-llm-integrating-dual-encoders-and","title":"Ideal-LLM: Integrating Dual Encoders and Language-Adapted LLM for Multilingual Speech-to-Text","date":"2024-09-17","arxiv_id":"2409.11214","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-best-rq-a-multi-channel-speech-foundation","title":"M-BEST-RQ: A Multi-Channel Speech Foundation Model for Smart Glasses","date":"2024-09-17","arxiv_id":"2409.11494","repositories_listed":0,"syntology":null},{"url":null,"slug":"wer-we-stand-benchmarking-urdu-asr-models","title":"WER We Stand: Benchmarking Urdu ASR Models","date":"2024-09-17","arxiv_id":"2409.11252","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-augmentation-for","title":"Zero Shot Text to Speech Augmentation for Automatic Speech Recognition on Low-Resource Accented Speech Corpora","date":"2024-09-17","arxiv_id":"2409.11107","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-zero-shot-non-intrusive-speech","title":"A Study on Zero-shot Non-intrusive Speech Assessment using Large Language Models","date":"2024-09-16","arxiv_id":"2409.09914","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-self-learning-framework-for","title":"An Efficient Self-Learning Framework For Interactive Spoken Dialog Systems","date":"2024-09-16","arxiv_id":"2409.10515","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-automatic-speech-recognition","title":"Augmenting Automatic Speech Recognition Models with Disfluency Detection","date":"2024-09-16","arxiv_id":"2409.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-whisper-speech-based-meta-icl-for-asr-on","title":"SMILE: Speech Meta In-Context Learning for Low-Resource Language Automatic Speech Recognition","date":"2024-09-16","arxiv_id":"2409.10429","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-based-generative-error","title":"Large Language Model Based Generative Error Correction: A Challenge and Baselines for Speech Recognition, Speaker Tagging, and Emotion Recognition","date":"2024-09-15","arxiv_id":"2409.09785","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-error-correction-using-large-language","title":"ASR Error Correction using Large Language Models","date":"2024-09-14","arxiv_id":"2409.09554","repositories_listed":0,"syntology":null},{"url":null,"slug":"clean-label-attacks-against-slu-systems","title":"Clean Label Attacks against SLU Systems","date":"2024-09-13","arxiv_id":"2409.08985","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpt-boosted-wav2vec2-0-towards-noise-robust","title":"CPT-Boosted Wav2vec2.0: Towards Noise Robust Speech Recognition for Classroom Environments","date":"2024-09-13","arxiv_id":"2409.14494","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-ssl-discrete-tokens-for","title":"Exploring SSL Discrete Tokens for Multilingual ASR","date":"2024-09-13","arxiv_id":"2409.08805","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-impact-of-data-quantity-on-asr","title":"Exploring the Impact of Data Quantity on ASR in Extremely Low-resource Languages","date":"2024-09-13","arxiv_id":"2409.08872","repositories_listed":0,"syntology":null},{"url":null,"slug":"la-rag-enhancing-llm-based-asr-accuracy-with","title":"LA-RAG:Enhancing LLM-based ASR Accuracy with Retrieval-Augmented Generation","date":"2024-09-13","arxiv_id":"2409.08597","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnings-from-curating-a-trustworthy-well","title":"Learnings from curating a trustworthy, well-annotated, and useful dataset of disordered English speech","date":"2024-09-13","arxiv_id":"2409.09190","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-speech-transformer-decoders-when","title":"Multi-modal Speech Transformer Decoders: When Do Multiple Modalities Improve Accuracy?","date":"2024-09-13","arxiv_id":"2409.09221","repositories_listed":0,"syntology":null},{"url":null,"slug":"nest-rq-next-token-prediction-for-speech-self","title":"NEST-RQ: Next Token Prediction for Speech Self-Supervised Pre-Training","date":"2024-09-13","arxiv_id":"2409.08680","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-landmark-acoustic-landmark-dataset-and","title":"Auto-Landmark: Acoustic Landmark Dataset and Open-Source Toolkit for Landmark Extraction","date":"2024-09-12","arxiv_id":"2409.07969","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-speech-llama-inference-with-multi","title":"Faster Speech-LLaMA Inference with Multi-token Prediction","date":"2024-09-12","arxiv_id":"2409.08148","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-text-error-correction-for-chinese-speech","title":"Full-text Error Correction for Chinese Speech Recognition with Large Language Model","date":"2024-09-12","arxiv_id":"2409.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-faetar-benchmark-speech-recognition-in-a","title":"The Faetar Benchmark: Speech Recognition in a Very Under-Resourced Language","date":"2024-09-12","arxiv_id":"2409.08103","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualization-of-asr-with-llm-using","title":"Contextualization of ASR with LLM using phonetic retrieval-based augmentation","date":"2024-09-11","arxiv_id":"2409.15353","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-ctc-based-visual-speech-recognition","title":"Enhancing CTC-Based Visual Speech Recognition","date":"2024-09-11","arxiv_id":"2409.07210","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-mamba-in-speech-processing-by-self","title":"Rethinking Mamba in Speech Processing by Self-Supervised Models","date":"2024-09-11","arxiv_id":"2409.07273","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-context-balanced-adaptation","title":"An Effective Context-Balanced Adaptation Approach for Long-Tailed Speech Recognition","date":"2024-09-10","arxiv_id":"2409.06468","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-redundant-is-the-transformer-stack-in","title":"How Redundant Is the Transformer Stack in Speech Representation Models?","date":"2024-09-10","arxiv_id":"2409.16302","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-aware-asr-error-augmentation-for","title":"Keyword-Aware ASR Error Augmentation for Robust Dialogue State Tracking","date":"2024-09-10","arxiv_id":"2409.06263","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-toolkit-for-joint-speaker-diarization-and","title":"A Toolkit for Joint Speaker Diarization and Identification with Application to Speaker-Attributed ASR","date":"2024-09-09","arxiv_id":"2409.05750","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-modularity-for-noise","title":"An investigation of modularity for noise robustness in conformer-based ASR","date":"2024-09-09","arxiv_id":"2409.05589","repositories_listed":0,"syntology":null},{"url":null,"slug":"consensus-based-distributed-quantum-kernel","title":"Consensus-based Distributed Quantum Kernel Learning for Speech Recognition","date":"2024-09-09","arxiv_id":"2409.05770","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-real-time-transcriptions-using","title":"Evaluation of real-time transcriptions using end-to-end ASR models","date":"2024-09-09","arxiv_id":"2409.05674","repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-2024-mandarin-stuttering","title":"Findings of the 2024 Mandarin Stuttering Event Detection and Automatic Speech Recognition Challenge","date":"2024-09-09","arxiv_id":"2409.05430","repositories_listed":0,"syntology":null},{"url":null,"slug":"longer-is-not-necessarily-stronger-punctuated","title":"Longer is (Not Necessarily) Stronger: Punctuated Long-Sequence Training for Enhanced Speech Recognition and Translation","date":"2024-09-09","arxiv_id":"2409.05601","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntt-multi-speaker-asr-system-for-the-dasr","title":"NTT Multi-Speaker ASR System for the DASR Task of CHiME-8 Challenge","date":"2024-09-09","arxiv_id":"2409.05554","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-correction-of-named","title":"Retrieval Augmented Correction of Named Entity Speech Recognition Errors","date":"2024-09-09","arxiv_id":"2409.06062","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-wavlm-back-ends-for-speech-spoofing","title":"Exploring WavLM Back-ends for Speech Spoofing and Deepfake Detection","date":"2024-09-08","arxiv_id":"2409.05032","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-extraction-of-noise-robust-discrete","title":"Efficient Extraction of Noise-Robust Discrete Units from Self-Supervised Speech Models","date":"2024-09-04","arxiv_id":"2409.02565","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-self-attention-in-self-supervised","title":"Probing self-attention in self-supervised speech models for cross-linguistic differences","date":"2024-09-04","arxiv_id":"2409.03115","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantification-of-stylistic-differences-in","title":"Quantification of stylistic differences in human- and ASR-produced transcripts of African American English","date":"2024-09-04","arxiv_id":"2409.03059","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-lost-in-normalization-exploring","title":"What is lost in Normalization? Exploring Pitfalls in Multilingual ASR Model Evaluations","date":"2024-09-04","arxiv_id":"2409.02449","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-code-switching-speech-recognition-1","title":"Enhancing Code-Switching Speech Recognition with LID-Based Collaborative Mixture of Experts Model","date":"2024-09-03","arxiv_id":"2409.02050","repositories_listed":0,"syntology":null},{"url":null,"slug":"reassessing-noise-augmentation-methods-in-the","title":"Reassessing Noise Augmentation Methods in the Context of Adversarial Speech","date":"2024-09-03","arxiv_id":"2409.01813","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-order-preserved-optimal-transport","title":"Temporal Order Preserved Optimal Transport-based Cross-modal Knowledge Transfer Learning for ASR","date":"2024-09-03","arxiv_id":"2409.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ustc-nercslip-systems-for-the-chime-8","title":"The USTC-NERCSLIP Systems for the CHiME-8 NOTSOFAR-1 Challenge","date":"2024-09-03","arxiv_id":"2409.02041","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxhakka-a-dialectally-diverse-multi-speaker","title":"VoxHakka: A Dialectally Diverse Multi-speaker Text-to-Speech System for Taiwanese Hakka","date":"2024-09-03","arxiv_id":"2409.01548","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-synthetic-audio-conversations","title":"A Framework for Synthetic Audio Conversations Generation using Large Language Models","date":"2024-09-02","arxiv_id":"2409.00946","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-adaptation-of-speech","title":"Resource-Efficient Adaptation of Speech Foundation Models for Multi-Speaker ASR","date":"2024-09-02","arxiv_id":"2409.01438","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-discrete-and-continuous-space-llms","title":"Comparing Discrete and Continuous Space LLMs for Speech Recognition","date":"2024-09-01","arxiv_id":"2409.00800","repositories_listed":0,"syntology":null},{"url":null,"slug":"serialized-speech-information-guidance-with","title":"Serialized Speech Information Guidance with Overlapped Encoding Separation for Multi-Speaker Automatic Speech Recognition","date":"2024-09-01","arxiv_id":"2409.00815","repositories_listed":0,"syntology":null},{"url":null,"slug":"dcim-avsr-efficient-audio-visual-speech","title":"DCIM-AVSR : Efficient Audio-Visual Speech Recognition via Dual Conformer Interaction Module","date":"2024-08-31","arxiv_id":"2409.00481","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-residual-extraction-based-pre","title":"Progressive Residual Extraction based Pre-training for Speech Representation Learning","date":"2024-08-31","arxiv_id":"2409.00387","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-multi-talker-asr-performance-with","title":"Advancing Multi-talker ASR Performance with Large Language Models","date":"2024-08-30","arxiv_id":"2408.17431","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-an-end-to-end-framework-for","title":"Developing an End-to-End Framework for Predicting the Social Communication Severity Scores of Children with Autism Spectrum Disorder","date":"2024-08-30","arxiv_id":"2409.00158","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-tagging-correction-with-non","title":"Speaker Tagging Correction With Non-Autoregressive Language Models","date":"2024-08-30","arxiv_id":"2409.00151","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-japanese-speech-recognition-on","title":"Benchmarking Japanese Speech Recognition on ASR-LLM Setups with Multi-Pass Augmented Generative Error Correction","date":"2024-08-29","arxiv_id":"2408.16180","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisit-micro-batch-clipping-adaptive-data","title":"Revisit Micro-batch Clipping: Adaptive Data Pruning via Gradient Manipulation","date":"2024-08-29","arxiv_id":"2408.16204","repositories_listed":0,"syntology":null},{"url":null,"slug":"literary-and-colloquial-dialect","title":"Literary and Colloquial Dialect Identification for Tamil using Acoustic Features","date":"2024-08-27","arxiv_id":"2408.14887","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-transformers-topological","title":"Speech Recognition Transformers: Topological-lingualism Perspective","date":"2024-08-27","arxiv_id":"2408.14991","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-recognition-and-detection-of","title":"Automatic recognition and detection of aphasic natural speech","date":"2024-08-26","arxiv_id":"2408.14082","repositories_listed":0,"syntology":null},{"url":null,"slug":"medsage-enhancing-robustness-of-medical","title":"MEDSAGE: Enhancing Robustness of Medical Dialogue Summarization to ASR Errors with LLM-generated Synthetic Dialogues","date":"2024-08-26","arxiv_id":"2408.14418","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-advances-and-new-paradigms-for","title":"Research Advances and New Paradigms for Biology-inspired Spiking Neural Networks","date":"2024-08-26","arxiv_id":"2408.13996","repositories_listed":0,"syntology":null},{"url":null,"slug":"literary-and-colloquial-tamil-dialect","title":"Literary and Colloquial Tamil Dialect Identification","date":"2024-08-25","arxiv_id":"2408.13739","repositories_listed":0,"syntology":null},{"url":null,"slug":"studying-the-effect-of-audio-filters-in-pre","title":"Studying the Effect of Audio Filters in Pre-Trained Models for Environmental Sound Classification","date":"2024-08-24","arxiv_id":"2408.13644","repositories_listed":0,"syntology":null},{"url":null,"slug":"focused-discriminative-training-for-streaming","title":"Focused Discriminative Training For Streaming CTC-Trained Automatic Speech Recognition Models","date":"2024-08-23","arxiv_id":"2408.13008","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-vocal-system-impaired-patient","title":"Developing vocal system impaired patient-aimed voice quality assessment approach using ASR representation-included multiple features","date":"2024-08-22","arxiv_id":"2408.12279","repositories_listed":0,"syntology":null},{"url":null,"slug":"positional-description-for-numerical","title":"Positional Description for Numerical Normalization","date":"2024-08-22","arxiv_id":"2408.12430","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-measuring-fairness-in-speech-1","title":"Towards measuring fairness in speech recognition: Fair-Speech dataset","date":"2024-08-22","arxiv_id":"2408.12734","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-error-prediction","title":"Improving Speech Recognition Error Prediction for Modern and Off-the-shelf Speech Recognizers","date":"2024-08-21","arxiv_id":"2408.11258","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-state-of-commercial-automatic-french","title":"The State of Commercial Automatic French Legal Speech Recognition Systems and their Impact on Court Reporters et al","date":"2024-08-21","arxiv_id":"2408.11940","repositories_listed":0,"syntology":null},{"url":null,"slug":"xcb-an-effective-contextual-biasing-approach","title":"XCB: an effective contextual biasing approach to bias cross-lingual phrases in speech recognition","date":"2024-08-20","arxiv_id":"2408.10524","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-transfer-learning-under","title":"Parameter-Efficient Transfer Learning under Federated Learning for Automatic Speech Recognition","date":"2024-08-19","arxiv_id":"2408.11873","repositories_listed":0,"syntology":null},{"url":null,"slug":"recording-for-eyes-not-echoing-to-ears","title":"Recording for Eyes, Not Echoing to Ears: Contextualized Spoken-to-Written Conversion of ASR Transcripts","date":"2024-08-19","arxiv_id":"2408.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-large-scale-spiking-neural-networks-a","title":"Toward Large-scale Spiking Neural Networks: A Comprehensive Survey and Future Directions","date":"2024-08-19","arxiv_id":"2409.02111","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-large-language-model-based-speech","title":"Enhancing Large Language Model-based Speech Recognition by Contextualization for Rare and Ambiguous Words","date":"2024-08-15","arxiv_id":"2408.08027","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpsnn-spiking-neural-network-for-low-latency","title":"DPSNN: Spiking Neural Network for Low-Latency Streaming Speech Enhancement","date":"2024-08-14","arxiv_id":"2408.07388","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-talker-finetuning-audio-language-model","title":"Style-Talker: Finetuning Audio Language Model and Style-Based Text-to-Speech Model for Fast Spoken Dialogue Generation","date":"2024-08-13","arxiv_id":"2408.11849","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-enhancement-for-computer-audition-an","title":"Audio Enhancement for Computer Audition -- An Iterative Training Paradigm Using Sample Importance","date":"2024-08-12","arxiv_id":"2408.06264","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-conversational-speech","title":"Cross-Lingual Conversational Speech Summarization with Large Language Models","date":"2024-08-12","arxiv_id":"2408.06484","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-dialogue-speech-recognition-with","title":"Enhancing Dialogue Speech Recognition with Robust Contextual Awareness via Noise Representation Learning","date":"2024-08-12","arxiv_id":"2408.06043","repositories_listed":0,"syntology":null},{"url":null,"slug":"vq-ctap-cross-modal-fine-grained-sequence","title":"VQ-CTAP: Cross-Modal Fine-Grained Sequence Representation Learning for Speech Processing","date":"2024-08-11","arxiv_id":"2408.05758","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-whisper-s-recognition-performance","title":"Improving Whisper's Recognition Performance for Under-Represented Language Kazakh Leveraging Unpaired Speech and Text","date":"2024-08-10","arxiv_id":"2408.05554","repositories_listed":0,"syntology":null},{"url":"/paper/mathbridge-a-large-scale-dataset-for","slug":"mathbridge-a-large-scale-dataset-for","title":"MathBridge: A Large Corpus Dataset for Translating Spoken Mathematical Expressions into $LaTeX$ Formulas for Improved Readability","date":"2024-08-07","arxiv_id":"2408.07081","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02945","title":"Self-Supervised Learning for Multi-Channel Neural Transducer","date":"2024-08-06","arxiv_id":"2408.02945","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02978","title":"ASR-enhanced Multimodal Representation Learning for Cross-Domain Product Retrieval","date":"2024-08-06","arxiv_id":"2408.02978","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamvoice-evolving-into-end-to-end","title":"StreamVoice+: Evolving into End-to-end Streaming Zero-shot Voice Conversion","date":"2024-08-05","arxiv_id":"2408.02178","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00205","title":"Sentence-wise Speech Summarization: Task, Datasets, and End-to-End Modeling with LM Knowledge Distillation","date":"2024-08-01","arxiv_id":"2408.00205","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21414","title":"Towards interfacing large language models with ASR systems using confidence measures and prompting","date":"2024-07-31","arxiv_id":"2407.21414","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21476","title":"On the Problem of Text-To-Speech Model Selection for Synthetic Data Generation in Automatic Speech Recognition","date":"2024-07-31","arxiv_id":"2407.21476","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21066","title":"ELP-Adapters: Parameter Efficient Adapter Tuning for Various Speech Processing Tasks","date":"2024-07-28","arxiv_id":"2407.21066","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21061","title":"Improving noisy student training for low-resource languages in End-to-End ASR using CycleGAN and inter-domain losses","date":"2024-07-26","arxiv_id":"2407.21061","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-bandwidth-expansion-via-high-fidelity","title":"Speech Bandwidth Expansion Via High Fidelity Generative Adversarial Networks","date":"2024-07-26","arxiv_id":"2407.18571","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-domain-specific-asr-with-llm","title":"Improving Domain-Specific ASR with LLM-Generated Contextual Descriptions","date":"2024-07-25","arxiv_id":"2407.17874","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effect-of-purely-synthetic-training","title":"On the Effect of Purely Synthetic Training Data for Different Automatic Speech Recognition Architectures","date":"2024-07-25","arxiv_id":"2407.17997","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-bilingual-and","title":"A Comparative Analysis of Bilingual and Trilingual Wav2Vec Models for Automatic Speech Recognition in Multilingual Oral History Archives","date":"2024-07-24","arxiv_id":"2407.17160","repositories_listed":0,"syntology":null},{"url":null,"slug":"coupling-speech-encoders-with-downstream-text","title":"Coupling Speech Encoders with Downstream Text Models","date":"2024-07-24","arxiv_id":"2407.17605","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-the-role-of-textual","title":"Quantifying the Role of Textual Predictability in Automatic Speech Recognition","date":"2024-07-23","arxiv_id":"2407.16537","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-chime-8-dasr-challenge-for-generalizable","title":"The CHiME-8 DASR Challenge for Generalizable and Array Agnostic Distant Automatic Speech Recognition and Diarization","date":"2024-07-23","arxiv_id":"2407.16447","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-speech-separation-models-for","title":"Robustness of Speech Separation Models for Similar-pitch Speakers","date":"2024-07-22","arxiv_id":"2407.15749","repositories_listed":0,"syntology":null},{"url":null,"slug":"trading-devil-final-backdoor-attack-via-stock","title":"Trading Devil Final: Backdoor attack via Stock market and Bayesian Optimization","date":"2024-07-21","arxiv_id":"2407.14573","repositories_listed":0,"syntology":null}],"record_sha256":"1952ed81c5596e7841acfa4510cf6a2c64f53c68dffab6f7c5e2f41a0477b3cf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}