{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/19","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":58,"rows_per_page":100,"rows":[1801,1900],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/18","next":"/task/speech-recognition-1/papers/20","papers":[{"url":null,"slug":"ge2e-ac-generalized-end-to-end-loss-training","title":"GE2E-AC: Generalized End-to-End Loss Training for Accent Classification","date":"2024-07-19","arxiv_id":"2407.14021","repositories_listed":0,"syntology":null},{"url":null,"slug":"reexamining-racial-disparities-in-automatic","title":"Reexamining Racial Disparities in Automatic Speech Recognition Performance: The Role of Confounding by Provenance","date":"2024-07-19","arxiv_id":"2407.13982","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00004","title":"Handling Numeric Expressions in Automatic Speech Recognition","date":"2024-07-18","arxiv_id":"2408.00004","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-light-weight-and-efficient-punctuation-and","title":"A light-weight and efficient punctuation and word casing prediction model for on-device streaming ASR","date":"2024-07-18","arxiv_id":"2407.13142","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resourced-speech-recognition-for-iu-mien","title":"Low-Resourced Speech Recognition for Iu Mien Language via Weakly-Supervised Phoneme-based Multilingual Pre-training","date":"2024-07-18","arxiv_id":"2407.13292","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-asr-error-correction-with-conservative","title":"Robust ASR Error Correction with Conservative Data Filtering","date":"2024-07-18","arxiv_id":"2407.13300","repositories_listed":0,"syntology":null},{"url":null,"slug":"morphosyntactic-analysis-for-childes","title":"Morphosyntactic Analysis for CHILDES","date":"2024-07-17","arxiv_id":"2407.12389","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-binary-multiclass-paraphasia-detection","title":"Beyond Binary: Multiclass Paraphasia Detection with Generative Pretrained Transformers and End-to-End Models","date":"2024-07-16","arxiv_id":"2407.11345","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-voiceprivacy-2022-challenge-progress-and","title":"The VoicePrivacy 2022 Challenge: Progress and Perspectives in Voice Anonymisation","date":"2024-07-16","arxiv_id":"2407.11516","repositories_listed":0,"syntology":null},{"url":null,"slug":"leave-no-knowledge-behind-during-knowledge","title":"Leave No Knowledge Behind During Knowledge Distillation: Towards Practical and Effective Knowledge Distillation for Code-Switching ASR Using Realistic Data","date":"2024-07-15","arxiv_id":"2407.10603","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-neural-biasing-for-contextual","title":"Improving Neural Biasing for Contextual Speech Recognition by Early Context Injection and Text Perturbation","date":"2024-07-14","arxiv_id":"2407.10303","repositories_listed":0,"syntology":null},{"url":null,"slug":"tamil-language-computing-the-present-and-the","title":"Tamil Language Computing: the Present and the Future","date":"2024-07-11","arxiv_id":"2407.08618","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-encoder-size-based-on-data-driven","title":"Dynamic Encoder Size Based on Data-Driven Layer-wise Pruning for Speech Recognition","date":"2024-07-10","arxiv_id":"2407.18930","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-voice-command-pipelines-for-drone","title":"Evaluating Voice Command Pipelines for Drone Control: From STT and LLM to Direct Classification and Siamese Networks","date":"2024-07-10","arxiv_id":"2407.08658","repositories_listed":0,"syntology":null},{"url":null,"slug":"hebdb-a-weakly-supervised-dataset-for-hebrew","title":"HebDB: a Weakly Supervised Dataset for Hebrew Speech Processing","date":"2024-07-10","arxiv_id":"2407.07566","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-speech-unit-selection-for-textless","title":"Analyzing Speech Unit Selection for Textless Speech-to-Speech Translation","date":"2024-07-08","arxiv_id":"2407.18332","repositories_listed":0,"syntology":null},{"url":null,"slug":"homogeneous-speaker-features-for-on-the-fly","title":"Homogeneous Speaker Features for On-the-Fly Dysarthric and Elderly Speaker Adaptation","date":"2024-07-08","arxiv_id":"2407.06310","repositories_listed":0,"syntology":null},{"url":null,"slug":"morse-code-enabled-speech-recognition-for","title":"Morse Code-Enabled Speech Recognition for Individuals with Visual and Hearing Impairments","date":"2024-07-07","arxiv_id":"2407.14525","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnervoice-a-dataset-of-non-native-english","title":"LearnerVoice: A Dataset of Non-Native English Learners' Spontaneous Speech","date":"2024-07-05","arxiv_id":"2407.04280","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitaper-mel-spectrograms-for-keyword","title":"Multitaper mel-spectrograms for keyword spotting","date":"2024-07-05","arxiv_id":"2407.04662","repositories_listed":0,"syntology":null},{"url":null,"slug":"romanization-encoding-for-multilingual-asr","title":"Romanization Encoding For Multilingual ASR","date":"2024-07-05","arxiv_id":"2407.04368","repositories_listed":0,"syntology":null},{"url":"/paper/seed-asr-understanding-diverse-speech-and","slug":"seed-asr-understanding-diverse-speech-and","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","date":"2024-07-05","arxiv_id":"2407.04675","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-for-code-switching","title":"Semi-supervised Learning for Code-Switching ASR with Large Language Model Filter","date":"2024-07-05","arxiv_id":"2407.04219","repositories_listed":0,"syntology":null},{"url":null,"slug":"speculative-speech-recognition-by-audio","title":"Speculative Speech Recognition by Audio-Prefixed Low-Rank Adaptation of Language Models","date":"2024-07-05","arxiv_id":"2407.04641","repositories_listed":0,"syntology":null},{"url":null,"slug":"xlsr-transducer-streaming-asr-for-self","title":"XLSR-Transducer: Streaming ASR for Self-Supervised Pretrained Models","date":"2024-07-05","arxiv_id":"2407.04439","repositories_listed":0,"syntology":null},{"url":null,"slug":"finetuning-end-to-end-models-for-estonian","title":"Finetuning End-to-End Models for Estonian Conversational Spoken Language Translation","date":"2024-07-04","arxiv_id":"2407.03809","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accented-speech-recognition-using","title":"Improving Accented Speech Recognition using Data Augmentation based on Unsupervised Text-to-Speech Synthesis","date":"2024-07-04","arxiv_id":"2407.04047","repositories_listed":0,"syntology":null},{"url":null,"slug":"serialized-output-training-by-learned","title":"Serialized Output Training by Learned Dominance","date":"2024-07-04","arxiv_id":"2407.03966","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-framework-for-animal-sound","title":"Advanced Framework for Animal Sound Classification With Features Optimization","date":"2024-07-03","arxiv_id":"2407.03440","repositories_listed":0,"syntology":null},{"url":null,"slug":"codec-asr-training-performant-automatic","title":"Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations","date":"2024-07-03","arxiv_id":"2407.03495","repositories_listed":0,"syntology":null},{"url":null,"slug":"qifusion-net-layer-adapted-stream-non-stream","title":"Qifusion-Net: Layer-adapted Stream/Non-stream Model for End-to-End Multi-Accent Speech Recognition","date":"2024-07-03","arxiv_id":"2407.03026","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-asr-models-and-features-for","title":"Self-supervised ASR Models and Features For Dysarthric and Elderly Speech Recognition","date":"2024-07-03","arxiv_id":"2407.13782","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ustc-nercslip-systems-for-the-icmc-asr","title":"The USTC-NERCSLIP Systems for The ICMC-ASR Challenge","date":"2024-07-02","arxiv_id":"2407.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-next-frontier-in-speech","title":"Towards the Next Frontier in Speech Representation Learning Using Disentanglement","date":"2024-07-02","arxiv_id":"2407.02543","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-transfer-learning-for-speech","title":"Cross-Lingual Transfer Learning for Speech Translation","date":"2024-07-01","arxiv_id":"2407.01130","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-automated-detection-of-biased-social","title":"Toward Automated Detection of Biased Social Signals from the Content of Clinical Conversations","date":"2024-07-01","arxiv_id":"2407.17477","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-forgetting-for-better-generalization","title":"Less Forgetting for Better Generalization: Exploring Continual-learning Fine-tuning Methods for Speech Self-supervised Representations","date":"2024-06-30","arxiv_id":"2407.00756","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-correction-by-paying-attention-to-both","title":"Error Correction by Paying Attention to Both Acoustic and Confidence References for Automatic Speech Recognition","date":"2024-06-29","arxiv_id":"2407.12817","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-source-conversational-ai-with","title":"Open-Source Conversational AI with SpeechBrain 1.0","date":"2024-06-29","arxiv_id":"2407.00463","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-is-more-accurate-speech-recognition","title":"Less is More: Accurate Speech Recognition & Translation without Web-Scale Data","date":"2024-06-28","arxiv_id":"2406.19674","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-llms-for-rescoring-n-best-asr","title":"Applying LLMs for Rescoring N-best ASR Hypotheses of Casual Conversations: Effects of Domain Adaptation and Context Carry-over","date":"2024-06-27","arxiv_id":"2406.18972","repositories_listed":0,"syntology":null},{"url":null,"slug":"tradition-or-innovation-a-comparison-of","title":"Tradition or Innovation: A Comparison of Modern ASR Methods for Forced Alignment","date":"2024-06-27","arxiv_id":"2406.19363","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-hindi","title":"Automatic Speech Recognition for Hindi","date":"2024-06-26","arxiv_id":"2406.18135","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-data-pruning-for-automatic-speech","title":"Dynamic Data Pruning for Automatic Speech Recognition","date":"2024-06-26","arxiv_id":"2406.18373","repositories_listed":0,"syntology":null},{"url":null,"slug":"msr-86k-an-evolving-multilingual-corpus-with","title":"MSR-86K: An Evolving, Multilingual Corpus with 86,300 Hours of Transcribed Audio for Speech Recognition Research","date":"2024-06-26","arxiv_id":"2406.18301","repositories_listed":0,"syntology":null},{"url":null,"slug":"sc-moe-switch-conformer-mixture-of-experts","title":"SC-MoE: Switch Conformer Mixture of Experts for Unified Streaming and Non-streaming Code-Switching ASR","date":"2024-06-26","arxiv_id":"2406.18021","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-solution-to-connect-speech","title":"A Comprehensive Solution to Connect Speech Encoder and Large Language Model for ASR","date":"2024-06-25","arxiv_id":"2406.17272","repositories_listed":0,"syntology":null},{"url":null,"slug":"msrs-training-multimodal-speech-recognition","title":"MSRS: Training Multimodal Speech Recognition Models from Scratch with Sparse Mask Optimization","date":"2024-06-25","arxiv_id":"2406.17614","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-editing-for-lifelong-training-of","title":"Sequential Editing for Lifelong Training of Speech Recognition Models","date":"2024-06-25","arxiv_id":"2406.17935","repositories_listed":0,"syntology":null},{"url":null,"slug":"blending-llms-into-cascaded-speech","title":"Blending LLMs into Cascaded Speech Translation: KIT's Offline Speech Translation System for IWSLT 2024","date":"2024-06-24","arxiv_id":"2406.16777","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-confidence-estimation-measures","title":"Investigating Confidence Estimation Measures for Speaker Diarization","date":"2024-06-24","arxiv_id":"2406.17124","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-end-to-end-automatic-speech","title":"Contextualized End-to-end Automatic Speech Recognition with Intermediate Biasing Loss","date":"2024-06-23","arxiv_id":"2406.16120","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoder-only-architecture-for-streaming-end","title":"Decoder-only Architecture for Streaming End-to-end Speech Recognition","date":"2024-06-23","arxiv_id":"2406.16107","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-feature-mixup-for-balanced-multi","title":"Acoustic Feature Mixup for Balanced Multi-aspect Pronunciation Assessment","date":"2024-06-22","arxiv_id":"2406.15723","repositories_listed":0,"syntology":null},{"url":null,"slug":"interbiasing-boost-unseen-word-recognition","title":"InterBiasing: Boost Unseen Word Recognition through Biasing Intermediate Predictions","date":"2024-06-21","arxiv_id":"2406.14890","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-of-phonological-assimilation-by","title":"Perception of Phonological Assimilation by Neural Speech Recognition Models","date":"2024-06-21","arxiv_id":"2406.15265","repositories_listed":0,"syntology":null},{"url":null,"slug":"pi-whisper-an-adaptive-and-incremental-asr","title":"PI-Whisper: Designing an Adaptive and Incremental Automatic Speech Recognition System for Edge Devices","date":"2024-06-21","arxiv_id":"2406.15668","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-adapter-based-unified-model-for-multiple","title":"An Adapter-Based Unified Model for Multiple Spoken Language Processing Tasks","date":"2024-06-20","arxiv_id":"2406.14747","repositories_listed":0,"syntology":null},{"url":null,"slug":"dasb-discrete-audio-and-speech-benchmark","title":"DASB -- Discrete Audio and Speech Benchmark","date":"2024-06-20","arxiv_id":"2406.14294","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-interface-enhancing-lecture","title":"Intelligent Interface: Enhancing Lecture Engagement with Didactic Activity Summaries","date":"2024-06-20","arxiv_id":"2406.14266","repositories_listed":0,"syntology":null},{"url":null,"slug":"children-s-speech-recognition-through","title":"Children's Speech Recognition through Discrete Token Enhancement","date":"2024-06-19","arxiv_id":"2406.13431","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-vs-sequential-speaker-role-detection","title":"Joint vs Sequential Speaker-Role Detection and Automatic Speech Recognition for Air-traffic Control","date":"2024-06-19","arxiv_id":"2406.13842","repositories_listed":0,"syntology":null},{"url":null,"slug":"manwav-the-first-manchu-asr-model","title":"ManWav: The First Manchu ASR Model","date":"2024-06-19","arxiv_id":"2406.13502","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-task-specific-subnetworks-in-multi","title":"Finding Task-specific Subnetworks in Multi-task Spoken Language Understanding Model","date":"2024-06-18","arxiv_id":"2406.12317","repositories_listed":0,"syntology":null},{"url":null,"slug":"performant-asr-models-for-medical-entities-in","title":"Performant ASR Models for Medical Entities in Accented Speech","date":"2024-06-18","arxiv_id":"2406.12387","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapid-language-adaptation-for-multilingual","title":"Rapid Language Adaptation for Multilingual E2E Speech Recognition Using Encoder Prompting","date":"2024-06-18","arxiv_id":"2406.12611","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribe-align-and-segment-creating-speech","title":"Transcribe, Align and Segment: Creating speech datasets for low-resource languages","date":"2024-06-18","arxiv_id":"2406.12674","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-biomedical","title":"Automatic Speech Recognition for Biomedical Data in Bengali Language","date":"2024-06-16","arxiv_id":"2406.12931","repositories_listed":0,"syntology":null},{"url":null,"slug":"costa-code-switched-speech-translation-using","title":"CoSTA: Code-Switched Speech Translation using Aligned Speech-Text Interleaving","date":"2024-06-16","arxiv_id":"2406.10993","repositories_listed":0,"syntology":null},{"url":null,"slug":"imperceptible-rhythm-backdoor-attacks","title":"Imperceptible Rhythm Backdoor Attacks: Exploring Rhythm Transformation for Embedding Undetectable Vulnerabilities on Speech Recognition","date":"2024-06-16","arxiv_id":"2406.10932","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-dysfluency","title":"Large Language Models for Dysfluency Detection in Stuttered Speech","date":"2024-06-16","arxiv_id":"2406.11025","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-emotion-recognition-using-cnn-and-its","title":"Speech Emotion Recognition Using CNN and Its Use Case in Digital Healthcare","date":"2024-06-15","arxiv_id":"2406.10741","repositories_listed":0,"syntology":null},{"url":null,"slug":"trading-devil-robust-backdoor-attack-via","title":"Trading Devil: Robust backdoor attack via Stochastic investment models and Bayesian approach","date":"2024-06-15","arxiv_id":"2406.10719","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-text-augmentation-approach-for","title":"An efficient text augmentation approach for contextualized Mandarin speech recognition","date":"2024-06-14","arxiv_id":"2406.09950","repositories_listed":0,"syntology":null},{"url":null,"slug":"cnvsrc-2023-the-first-chinese-continuous","title":"CNVSRC 2023: The First Chinese Continuous Visual Speech Recognition Challenge","date":"2024-06-14","arxiv_id":"2406.10313","repositories_listed":0,"syntology":null},{"url":null,"slug":"inclusive-asr-for-disfluent-speech-cascaded","title":"Inclusive ASR for Disfluent Speech: Cascaded Large-Scale Self-Supervised Learning with Targeted Fine-Tuning and Data Augmentation","date":"2024-06-14","arxiv_id":"2406.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-language-structures-through","title":"Learning Language Structures through Grounding","date":"2024-06-14","arxiv_id":"2406.09662","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-evaluation-of-speech-foundation-models","title":"On the Evaluation of Speech Foundation Models for Spoken Language Understanding","date":"2024-06-14","arxiv_id":"2406.10083","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-byte-level-representation-for-end","title":"Optimizing Byte-level Representation for End-to-end ASR","date":"2024-06-14","arxiv_id":"2406.09676","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceiver-prompt-flexible-speaker-adaptation","title":"Perceiver-Prompt: Flexible Speaker Adaptation in Whisper for Chinese Disordered Speech Recognition","date":"2024-06-14","arxiv_id":"2406.09873","repositories_listed":0,"syntology":null},{"url":null,"slug":"roar-reinforcing-original-to-augmented-data","title":"ROAR: Reinforcing Original to Augmented Data Ratio Dynamics for Wav2Vec2.0 Based ASR","date":"2024-06-14","arxiv_id":"2406.09999","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptwin-low-cost-adaptive-compression-of","title":"AdaPTwin: Low-Cost Adaptive Compression of Product Twins in Transformers","date":"2024-06-13","arxiv_id":"2406.08904","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-spoken-language-identification","title":"Exploring Spoken Language Identification Strategies for Automatic Transcription of Multilingual Broadcast and Institutional Speech","date":"2024-06-13","arxiv_id":"2406.09290","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-multi-speaker-asr-using-target","title":"Multi-Channel Multi-Speaker ASR Using Target Speaker's Solo Segment","date":"2024-06-13","arxiv_id":"2406.09589","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-retrieval-for-large-language","title":"Multi-Modal Retrieval For Large Language Model Based Speech Recognition","date":"2024-06-13","arxiv_id":"2406.09618","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-reallm-real-time-streaming-speech","title":"Speech ReaLLM -- Real-time Streaming Speech Recognition with Multimodal LLMs by Teaching the Flow of Time","date":"2024-06-13","arxiv_id":"2406.09569","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-second-displace-challenge-diarization-of","title":"The Second DISPLACE Challenge : DIarization of SPeaker and LAnguage in Conversational Environments","date":"2024-06-13","arxiv_id":"2406.09494","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcription-free-fine-tuning-of-speech","title":"Transcription-Free Fine-Tuning of Speech Separation Models for Noisy and Reverberant Multi-Speaker Automatic Speech Recognition","date":"2024-06-13","arxiv_id":"2406.08914","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-conditioned-phonemic-and-prosodic","title":"Audio-conditioned phonemic and prosodic annotation for building text-to-speech models from unlabeled speech data","date":"2024-06-12","arxiv_id":"2406.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-personalized-voice","title":"Comparative Analysis of Personalized Voice Activity Detection Systems: Assessing Real-World Effectiveness","date":"2024-06-12","arxiv_id":"2406.09443","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-pipeline-with-low-rank-adaptation-for","title":"Dual-Pipeline with Low-Rank Adaptation for New Language Integration in Multilingual ASR","date":"2024-06-12","arxiv_id":"2406.07842","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualvc-3-leveraging-language-model-generated","title":"DualVC 3: Leveraging Language Model Generated Pseudo Context for End-to-end Low Latency Streaming Voice Conversion","date":"2024-06-12","arxiv_id":"2406.07846","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-child-speech-recognition-with","title":"Improving child speech recognition with augmented child-like speech","date":"2024-06-12","arxiv_id":"2406.10284","repositories_listed":0,"syntology":null},{"url":null,"slug":"ml-superb-2-0-benchmarking-multilingual","title":"ML-SUPERB 2.0: Benchmarking Multilingual Speech Models Across Modeling Constraints, Languages, and Datasets","date":"2024-06-12","arxiv_id":"2406.08641","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-blind-source-separation-and","title":"Neural Blind Source Separation and Diarization for Distant Speech Recognition","date":"2024-06-12","arxiv_id":"2406.08396","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyspeech-exploring-unified-multitask-speech","title":"PolySpeech: Exploring Unified Multitask Speech Models for Competitiveness with Single-task Models","date":"2024-06-12","arxiv_id":"2406.07801","repositories_listed":0,"syntology":null},{"url":null,"slug":"prodeliberation-parallel-robust-deliberation","title":"PRoDeliberation: Parallel Robust Deliberation for End-to-End Spoken Language Understanding","date":"2024-06-12","arxiv_id":"2406.07823","repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-self-supervised-learnt-speech","title":"Refining Self-Supervised Learnt Speech Representation using Brain Activations","date":"2024-06-12","arxiv_id":"2406.08266","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-model-for-asr-n-best","title":"Transformer-based Model for ASR N-Best Rescoring and Rewriting","date":"2024-06-12","arxiv_id":"2406.08207","repositories_listed":0,"syntology":null},{"url":null,"slug":"as-70-a-mandarin-stuttered-speech-dataset-for","title":"AS-70: A Mandarin stuttered speech dataset for automatic speech recognition and stuttering event detection","date":"2024-06-11","arxiv_id":"2406.07256","repositories_listed":0,"syntology":null}],"record_sha256":"e3f58e93f732dd999fd355ee0279cf1378ae4b1d01f9f19abd05f5ddc1165d2b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}