{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/10","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":31,"rows_per_page":100,"rows":[901,1000],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/9","next":"/task/automatic-speech-recognition/papers/11","papers":[{"url":null,"slug":"leave-no-knowledge-behind-during-knowledge","title":"Leave No Knowledge Behind During Knowledge Distillation: Towards Practical and Effective Knowledge Distillation for Code-Switching ASR Using Realistic Data","date":"2024-07-15","arxiv_id":"2407.10603","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-neural-biasing-for-contextual","title":"Improving Neural Biasing for Contextual Speech Recognition by Early Context Injection and Text Perturbation","date":"2024-07-14","arxiv_id":"2407.10303","repositories_listed":0,"syntology":null},{"url":null,"slug":"hebdb-a-weakly-supervised-dataset-for-hebrew","title":"HebDB: a Weakly Supervised Dataset for Hebrew Speech Processing","date":"2024-07-10","arxiv_id":"2407.07566","repositories_listed":0,"syntology":null},{"url":null,"slug":"homogeneous-speaker-features-for-on-the-fly","title":"Homogeneous Speaker Features for On-the-Fly Dysarthric and Elderly Speaker Adaptation","date":"2024-07-08","arxiv_id":"2407.06310","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnervoice-a-dataset-of-non-native-english","title":"LearnerVoice: A Dataset of Non-Native English Learners' Spontaneous Speech","date":"2024-07-05","arxiv_id":"2407.04280","repositories_listed":0,"syntology":null},{"url":null,"slug":"romanization-encoding-for-multilingual-asr","title":"Romanization Encoding For Multilingual ASR","date":"2024-07-05","arxiv_id":"2407.04368","repositories_listed":0,"syntology":null},{"url":"/paper/seed-asr-understanding-diverse-speech-and","slug":"seed-asr-understanding-diverse-speech-and","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","date":"2024-07-05","arxiv_id":"2407.04675","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-for-code-switching","title":"Semi-supervised Learning for Code-Switching ASR with Large Language Model Filter","date":"2024-07-05","arxiv_id":"2407.04219","repositories_listed":0,"syntology":null},{"url":null,"slug":"speculative-speech-recognition-by-audio","title":"Speculative Speech Recognition by Audio-Prefixed Low-Rank Adaptation of Language Models","date":"2024-07-05","arxiv_id":"2407.04641","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accented-speech-recognition-using","title":"Improving Accented Speech Recognition using Data Augmentation based on Unsupervised Text-to-Speech Synthesis","date":"2024-07-04","arxiv_id":"2407.04047","repositories_listed":0,"syntology":null},{"url":null,"slug":"codec-asr-training-performant-automatic","title":"Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations","date":"2024-07-03","arxiv_id":"2407.03495","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-correction-by-paying-attention-to-both","title":"Error Correction by Paying Attention to Both Acoustic and Confidence References for Automatic Speech Recognition","date":"2024-06-29","arxiv_id":"2407.12817","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-llms-for-rescoring-n-best-asr","title":"Applying LLMs for Rescoring N-best ASR Hypotheses of Casual Conversations: Effects of Domain Adaptation and Context Carry-over","date":"2024-06-27","arxiv_id":"2406.18972","repositories_listed":0,"syntology":null},{"url":null,"slug":"tradition-or-innovation-a-comparison-of","title":"Tradition or Innovation: A Comparison of Modern ASR Methods for Forced Alignment","date":"2024-06-27","arxiv_id":"2406.19363","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-hindi","title":"Automatic Speech Recognition for Hindi","date":"2024-06-26","arxiv_id":"2406.18135","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-data-pruning-for-automatic-speech","title":"Dynamic Data Pruning for Automatic Speech Recognition","date":"2024-06-26","arxiv_id":"2406.18373","repositories_listed":0,"syntology":null},{"url":null,"slug":"msr-86k-an-evolving-multilingual-corpus-with","title":"MSR-86K: An Evolving, Multilingual Corpus with 86,300 Hours of Transcribed Audio for Speech Recognition Research","date":"2024-06-26","arxiv_id":"2406.18301","repositories_listed":0,"syntology":null},{"url":null,"slug":"sc-moe-switch-conformer-mixture-of-experts","title":"SC-MoE: Switch Conformer Mixture of Experts for Unified Streaming and Non-streaming Code-Switching ASR","date":"2024-06-26","arxiv_id":"2406.18021","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-editing-for-lifelong-training-of","title":"Sequential Editing for Lifelong Training of Speech Recognition Models","date":"2024-06-25","arxiv_id":"2406.17935","repositories_listed":0,"syntology":null},{"url":null,"slug":"blending-llms-into-cascaded-speech","title":"Blending LLMs into Cascaded Speech Translation: KIT's Offline Speech Translation System for IWSLT 2024","date":"2024-06-24","arxiv_id":"2406.16777","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoder-only-architecture-for-streaming-end","title":"Decoder-only Architecture for Streaming End-to-end Speech Recognition","date":"2024-06-23","arxiv_id":"2406.16107","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-of-phonological-assimilation-by","title":"Perception of Phonological Assimilation by Neural Speech Recognition Models","date":"2024-06-21","arxiv_id":"2406.15265","repositories_listed":0,"syntology":null},{"url":null,"slug":"pi-whisper-an-adaptive-and-incremental-asr","title":"PI-Whisper: Designing an Adaptive and Incremental Automatic Speech Recognition System for Edge Devices","date":"2024-06-21","arxiv_id":"2406.15668","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-vs-sequential-speaker-role-detection","title":"Joint vs Sequential Speaker-Role Detection and Automatic Speech Recognition for Air-traffic Control","date":"2024-06-19","arxiv_id":"2406.13842","repositories_listed":0,"syntology":null},{"url":null,"slug":"manwav-the-first-manchu-asr-model","title":"ManWav: The First Manchu ASR Model","date":"2024-06-19","arxiv_id":"2406.13502","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-task-specific-subnetworks-in-multi","title":"Finding Task-specific Subnetworks in Multi-task Spoken Language Understanding Model","date":"2024-06-18","arxiv_id":"2406.12317","repositories_listed":0,"syntology":null},{"url":null,"slug":"performant-asr-models-for-medical-entities-in","title":"Performant ASR Models for Medical Entities in Accented Speech","date":"2024-06-18","arxiv_id":"2406.12387","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribe-align-and-segment-creating-speech","title":"Transcribe, Align and Segment: Creating speech datasets for low-resource languages","date":"2024-06-18","arxiv_id":"2406.12674","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-biomedical","title":"Automatic Speech Recognition for Biomedical Data in Bengali Language","date":"2024-06-16","arxiv_id":"2406.12931","repositories_listed":0,"syntology":null},{"url":null,"slug":"costa-code-switched-speech-translation-using","title":"CoSTA: Code-Switched Speech Translation using Aligned Speech-Text Interleaving","date":"2024-06-16","arxiv_id":"2406.10993","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-text-augmentation-approach-for","title":"An efficient text augmentation approach for contextualized Mandarin speech recognition","date":"2024-06-14","arxiv_id":"2406.09950","repositories_listed":0,"syntology":null},{"url":null,"slug":"inclusive-asr-for-disfluent-speech-cascaded","title":"Inclusive ASR for Disfluent Speech: Cascaded Large-Scale Self-Supervised Learning with Targeted Fine-Tuning and Data Augmentation","date":"2024-06-14","arxiv_id":"2406.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-byte-level-representation-for-end","title":"Optimizing Byte-level Representation for End-to-end ASR","date":"2024-06-14","arxiv_id":"2406.09676","repositories_listed":0,"syntology":null},{"url":null,"slug":"roar-reinforcing-original-to-augmented-data","title":"ROAR: Reinforcing Original to Augmented Data Ratio Dynamics for Wav2Vec2.0 Based ASR","date":"2024-06-14","arxiv_id":"2406.09999","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-multi-speaker-asr-using-target","title":"Multi-Channel Multi-Speaker ASR Using Target Speaker's Solo Segment","date":"2024-06-13","arxiv_id":"2406.09589","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-second-displace-challenge-diarization-of","title":"The Second DISPLACE Challenge : DIarization of SPeaker and LAnguage in Conversational Environments","date":"2024-06-13","arxiv_id":"2406.09494","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcription-free-fine-tuning-of-speech","title":"Transcription-Free Fine-Tuning of Speech Separation Models for Noisy and Reverberant Multi-Speaker Automatic Speech Recognition","date":"2024-06-13","arxiv_id":"2406.08914","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-conditioned-phonemic-and-prosodic","title":"Audio-conditioned phonemic and prosodic annotation for building text-to-speech models from unlabeled speech data","date":"2024-06-12","arxiv_id":"2406.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualvc-3-leveraging-language-model-generated","title":"DualVC 3: Leveraging Language Model Generated Pseudo Context for End-to-end Low Latency Streaming Voice Conversion","date":"2024-06-12","arxiv_id":"2406.07846","repositories_listed":0,"syntology":null},{"url":null,"slug":"ml-superb-2-0-benchmarking-multilingual","title":"ML-SUPERB 2.0: Benchmarking Multilingual Speech Models Across Modeling Constraints, Languages, and Datasets","date":"2024-06-12","arxiv_id":"2406.08641","repositories_listed":0,"syntology":null},{"url":null,"slug":"prodeliberation-parallel-robust-deliberation","title":"PRoDeliberation: Parallel Robust Deliberation for End-to-End Spoken Language Understanding","date":"2024-06-12","arxiv_id":"2406.07823","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-model-for-asr-n-best","title":"Transformer-based Model for ASR N-Best Rescoring and Rewriting","date":"2024-06-12","arxiv_id":"2406.08207","repositories_listed":0,"syntology":null},{"url":null,"slug":"as-70-a-mandarin-stuttered-speech-dataset-for","title":"AS-70: A Mandarin stuttered speech dataset for automatic speech recognition and stuttering event detection","date":"2024-06-11","arxiv_id":"2406.07256","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-context-biasing-for-ctc-and-transducer","title":"Fast Context-Biasing for CTC and Transducer ASR models with CTC-based Word Spotter","date":"2024-06-11","arxiv_id":"2406.07096","repositories_listed":0,"syntology":null},{"url":null,"slug":"reading-miscue-detection-in-primary-school","title":"Reading Miscue Detection in Primary School through Automatic Speech Recognition","date":"2024-06-11","arxiv_id":"2406.07060","repositories_listed":0,"syntology":null},{"url":null,"slug":"astra-aligning-speech-and-text","title":"ASTRA: Aligning Speech and Text Representations for Asr without Sampling","date":"2024-06-10","arxiv_id":"2406.06664","repositories_listed":0,"syntology":null},{"url":null,"slug":"ms-hubert-mitigating-pre-training-and","title":"MS-HuBERT: Mitigating Pre-training and Inference Mismatch in Masked Language Modelling methods for learning Speech Representations","date":"2024-06-09","arxiv_id":"2406.05661","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-whisper-parameter-efficient-and","title":"LoRA-Whisper: Parameter-Efficient and Extensible Multilingual ASR","date":"2024-06-07","arxiv_id":"2406.06619","repositories_listed":0,"syntology":null},{"url":null,"slug":"pitch-aware-rnn-t-for-mandarin-chinese","title":"Pitch-Aware RNN-T for Mandarin Chinese Mispronunciation Detection and Diagnosis","date":"2024-06-07","arxiv_id":"2406.04595","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-multichannel-speech-enhancement-for","title":"Flexible Multichannel Speech Enhancement for Noise-Robust Frontend","date":"2024-06-06","arxiv_id":"2406.04552","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypernetworks-for-personalizing-asr-to","title":"Hypernetworks for Personalizing ASR to Atypical Speech","date":"2024-06-06","arxiv_id":"2406.04240","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-zero-shot-chinese-english-code","title":"Improving Zero-Shot Chinese-English Code-Switching ASR with kNN-CTC and Gated Monolingual Datastores","date":"2024-06-06","arxiv_id":"2406.03814","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-ctc-based-speech-recognition-with","title":"Enhancing CTC-based speech recognition with diverse modeling units","date":"2024-06-05","arxiv_id":"2406.03274","repositories_listed":0,"syntology":null},{"url":null,"slug":"syn2real-leveraging-task-arithmetic-for","title":"Task Arithmetic can Mitigate Synthetic-to-Real Gap in Automatic Speech Recognition","date":"2024-06-05","arxiv_id":"2406.02925","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-injection-for-neural-contextual-biasing","title":"Text Injection for Neural Contextual Biasing","date":"2024-06-05","arxiv_id":"2406.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiently-train-asr-models-that-memorize","title":"Efficiently Train ASR Models that Memorize Less and Perform Better with Per-core Clipping","date":"2024-06-04","arxiv_id":"2406.02004","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-guided-adaptation-of-automatic-speech","title":"Keyword-Guided Adaptation of Automatic Speech Recognition","date":"2024-06-04","arxiv_id":"2406.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-asr-for-low-resource-languages-a","title":"Enabling ASR for Low-Resource Languages: A Comprehensive Dataset Creation Approach","date":"2024-06-03","arxiv_id":"2406.01446","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2prompt-end-to-end-speech-prompt","title":"Wav2Prompt: End-to-End Speech Prompt Generation and Tuning For LLM in Zero and Few-shot Learning","date":"2024-06-01","arxiv_id":"2406.00522","repositories_listed":0,"syntology":null},{"url":null,"slug":"zipper-a-multi-tower-decoder-architecture-for","title":"Zipper: A Multi-Tower Decoder Architecture for Fusing Modalities","date":"2024-05-29","arxiv_id":"2405.18669","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-clinical-documentation-harnessing","title":"Intelligent Clinical Documentation: Harnessing Generative AI for Patient-Centric Clinical Note Generation","date":"2024-05-28","arxiv_id":"2405.18346","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-lm-pushing-the-limits-of-error","title":"Denoising LM: Pushing the Limits of Error Correction Models for Speech Recognition","date":"2024-05-24","arxiv_id":"2405.15216","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-optimization-of-streaming-and-non","title":"Joint Optimization of Streaming and Non-Streaming Automatic Speech Recognition with Multi-Decoder and Knowledge Distillation","date":"2024-05-22","arxiv_id":"2405.13514","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-don-t-understand-me-comparing-asr-results","title":"You don't understand me!: Comparing ASR results for L1 and L2 speakers of Swedish","date":"2024-05-22","arxiv_id":"2405.13379","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairlens-assessing-fairness-in-law","title":"FairLENS: Assessing Fairness in Law Enforcement Speech Recognition","date":"2024-05-21","arxiv_id":"2405.13166","repositories_listed":0,"syntology":null},{"url":null,"slug":"listen-again-and-choose-the-right-answer-a","title":"Listen Again and Choose the Right Answer: A New Paradigm for Automatic Speech Recognition with Large Language Models","date":"2024-05-16","arxiv_id":"2405.10025","repositories_listed":0,"syntology":null},{"url":null,"slug":"continued-pretraining-for-domain-adaptation","title":"Continued Pretraining for Domain Adaptation of Wav2vec2.0 in Automatic Speech Recognition for Elementary Math Classroom Settings","date":"2024-05-15","arxiv_id":"2405.13018","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-evaluating-the-robustness-of","title":"Towards Evaluating the Robustness of Automatic Speech Recognition Systems via Audio Style Transfer","date":"2024-05-15","arxiv_id":"2405.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"lost-in-transcription-identifying-and","title":"Lost in Transcription: Identifying and Quantifying the Accuracy Biases of Automatic Speech Recognition Systems Against Disfluent Speech","date":"2024-05-10","arxiv_id":"2405.06150","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmger-multi-modal-and-multi-granularity","title":"MMGER: Multi-modal and Multi-granularity Generative Error Correction with LLM for Joint Accent and Speech Recognition","date":"2024-05-06","arxiv_id":"2405.03152","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-x-vectors-and-bayesian-batch-active","title":"Combining X-Vectors and Bayesian Batch Active Learning: Two-Stage Active Learning Pipeline for Speech Recognition","date":"2024-05-03","arxiv_id":"2406.02566","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-compression-of-multitask","title":"Efficient Compression of Multitask Multilingual Speech Models","date":"2024-05-02","arxiv_id":"2405.00966","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-membership-inference-in-asr-model","title":"Improving Membership Inference in ASR Model Auditing with Perturbed Loss Features","date":"2024-05-02","arxiv_id":"2405.01207","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-to-sequence-models-in-peer-to-peer","title":"Sequence-to-sequence models in peer-to-peer learning: A practical application","date":"2024-05-02","arxiv_id":"2406.02565","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-whisper-understand-swiss-german-an","title":"Does Whisper understand Swiss German? An automatic, qualitative, and human evaluation","date":"2024-04-30","arxiv_id":"2404.19310","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-system","title":"Automatic Speech Recognition System-Independent Word Error Rate Estimation","date":"2024-04-25","arxiv_id":"2404.16743","repositories_listed":0,"syntology":null},{"url":null,"slug":"u2-moe-scaling-4-7x-parameters-with-minimal","title":"U2++ MoE: Scaling 4.7x parameters with minimal impact on RTF","date":"2024-04-25","arxiv_id":"2404.16407","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-walls-pioneering-automatic-speech","title":"Breaking Walls: Pioneering Automatic Speech Recognition for Central Kurdish: End-to-End Transformer Paradigm","date":"2024-04-23","arxiv_id":"2406.02561","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-processing-distortions","title":"Rethinking Processing Distortions: Disentangling the Impact of Speech Enhancement Errors on Speech Recognition Performance","date":"2024-04-23","arxiv_id":"2404.14860","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-neural-networks-to-recognize","title":"Artificial Neural Networks to Recognize Speakers Division from Continuous Bengali Speech","date":"2024-04-18","arxiv_id":"2404.15168","repositories_listed":0,"syntology":null},{"url":null,"slug":"anatomy-of-industrial-scale-multilingual-asr","title":"Anatomy of Industrial Scale Multilingual ASR","date":"2024-04-15","arxiv_id":"2404.09841","repositories_listed":0,"syntology":null},{"url":"/paper/asr-advancements-for-indigenous-languages","slug":"asr-advancements-for-indigenous-languages","title":"Automatic Speech Recognition Advancements for Indigenous Languages of the Americas","date":"2024-04-12","arxiv_id":"2404.08368","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asr-advancements-for-indigenous-languages#ran","syntology_url":"https://syntology.ai/paper/2404.08368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08368"}},"official":null}},{"url":null,"slug":"comparing-apples-to-oranges-llm-powered","title":"Comparing Apples to Oranges: LLM-powered Multimodal Intention Prediction in an Object Categorization Task","date":"2024-04-12","arxiv_id":"2404.08424","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-automated-speaking-assessment","title":"An Effective Automated Speaking Assessment Approach to Mitigating Data Scarcity and Imbalanced Distribution","date":"2024-04-11","arxiv_id":"2404.07575","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformer-1-robust-asr-via-large-scale","title":"Conformer-1: Robust ASR via Large-Scale Semisupervised Bootstrapping","date":"2024-04-10","arxiv_id":"2404.07341","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-x-lance-technical-report-for-interspeech","title":"The X-LANCE Technical Report for Interspeech 2024 Speech Processing Using Discrete Speech Unit Challenge","date":"2024-04-09","arxiv_id":"2404.06079","repositories_listed":0,"syntology":null},{"url":null,"slug":"mai-ho-omauna-i-ka-ai-language-models-improve","title":"Mai Ho'omāuna i ka 'Ai: Language Models Improve Automatic Speech Recognition in Hawaiian","date":"2024-04-03","arxiv_id":"2404.03073","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-masking-attacks-and-defenses-for","title":"Noise Masking Attacks and Defenses for Pretrained Speech Models","date":"2024-04-02","arxiv_id":"2404.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"houston-we-have-a-divergence-a-subgroup","title":"Houston we have a Divergence: A Subgroup Performance Analysis of ASR Models","date":"2024-03-31","arxiv_id":"2404.07226","repositories_listed":0,"syntology":null},{"url":null,"slug":"lv-ctc-non-autoregressive-asr-with-ctc-and","title":"LV-CTC: Non-autoregressive ASR with CTC and latent variable models","date":"2024-03-28","arxiv_id":"2403.19207","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-multi-modal-pre-training-for","title":"Multi-Stage Multi-Modal Pre-Training for Automatic Speech Recognition","date":"2024-03-28","arxiv_id":"2403.19822","repositories_listed":0,"syntology":null},{"url":null,"slug":"zaebuc-spoken-a-multilingual-multidialectal","title":"ZAEBUC-Spoken: A Multilingual Multidialectal Arabic-English Speech Corpus","date":"2024-03-27","arxiv_id":"2403.18182","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-biomedical-entities-from-noisy","title":"Extracting Biomedical Entities from Noisy Audio Transcripts","date":"2024-03-26","arxiv_id":"2403.17363","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-approach-to-device-directed","title":"A Multimodal Approach to Device-Directed Speech Detection with Large Language Models","date":"2024-03-21","arxiv_id":"2403.14438","repositories_listed":0,"syntology":null},{"url":null,"slug":"banglanum-a-public-dataset-for-bengali-digit","title":"BanglaNum -- A Public Dataset for Bengali Digit Recognition from Speech","date":"2024-03-20","arxiv_id":"2403.13465","repositories_listed":0,"syntology":null},{"url":null,"slug":"isometric-neural-machine-translation-using","title":"Isometric Neural Machine Translation using Phoneme Count Ratio Reward-based Reinforcement Learning","date":"2024-03-20","arxiv_id":"2403.15469","repositories_listed":0,"syntology":null},{"url":null,"slug":"adamer-ctc-connectionist-temporal","title":"AdaMER-CTC: Connectionist Temporal Classification with Adaptive Maximum Entropy Regularization for Automatic Speech Recognition","date":"2024-03-18","arxiv_id":"2403.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-artificial-intelligence-algorithms","title":"Artificial Intelligence for Cochlear Implants: Review of Strategies, Challenges, and Perspectives","date":"2024-03-17","arxiv_id":"2403.15442","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-asr-for-the","title":"Automatic Speech Recognition (ASR) for the Diagnosis of pronunciation of Speech Sound Disorders in Korean children","date":"2024-03-13","arxiv_id":"2403.08187","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-evaluation-of-a-code-switched-sepedi","title":"The evaluation of a code-switched Sepedi-English automatic speech recognition system","date":"2024-03-11","arxiv_id":"2403.07947","repositories_listed":0,"syntology":null}],"record_sha256":"9fa3e3d0328e8296eb014fee3b2a6efd364822d0d8ca2b8a41844c13761fbd9e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}