{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/12","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":32,"rows_per_page":100,"rows":[1101,1200],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/11","next":"/task/automatic-speech-recognition-2/papers/13","papers":[{"url":null,"slug":"improving-zero-shot-chinese-english-code","title":"Improving Zero-Shot Chinese-English Code-Switching ASR with kNN-CTC and Gated Monolingual Datastores","date":"2024-06-06","arxiv_id":"2406.03814","repositories_listed":0,"syntology":null},{"url":null,"slug":"4d-asr-joint-beam-search-integrating-ctc","title":"Joint Beam Search Integrating CTC, Attention, and Transducer Decoders","date":"2024-06-05","arxiv_id":"2406.02950","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-ctc-based-speech-recognition-with","title":"Enhancing CTC-based speech recognition with diverse modeling units","date":"2024-06-05","arxiv_id":"2406.03274","repositories_listed":0,"syntology":null},{"url":null,"slug":"syn2real-leveraging-task-arithmetic-for","title":"Task Arithmetic can Mitigate Synthetic-to-Real Gap in Automatic Speech Recognition","date":"2024-06-05","arxiv_id":"2406.02925","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-injection-for-neural-contextual-biasing","title":"Text Injection for Neural Contextual Biasing","date":"2024-06-05","arxiv_id":"2406.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiently-train-asr-models-that-memorize","title":"Efficiently Train ASR Models that Memorize Less and Perform Better with Per-core Clipping","date":"2024-06-04","arxiv_id":"2406.02004","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-guided-adaptation-of-automatic-speech","title":"Keyword-Guided Adaptation of Automatic Speech Recognition","date":"2024-06-04","arxiv_id":"2406.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-asr-for-low-resource-languages-a","title":"Enabling ASR for Low-Resource Languages: A Comprehensive Dataset Creation Approach","date":"2024-06-03","arxiv_id":"2406.01446","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2prompt-end-to-end-speech-prompt","title":"Wav2Prompt: End-to-End Speech Prompt Generation and Tuning For LLM in Zero and Few-shot Learning","date":"2024-06-01","arxiv_id":"2406.00522","repositories_listed":0,"syntology":null},{"url":null,"slug":"zipper-a-multi-tower-decoder-architecture-for","title":"Zipper: A Multi-Tower Decoder Architecture for Fusing Modalities","date":"2024-05-29","arxiv_id":"2405.18669","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-clinical-documentation-harnessing","title":"Intelligent Clinical Documentation: Harnessing Generative AI for Patient-Centric Clinical Note Generation","date":"2024-05-28","arxiv_id":"2405.18346","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-lm-pushing-the-limits-of-error","title":"Denoising LM: Pushing the Limits of Error Correction Models for Speech Recognition","date":"2024-05-24","arxiv_id":"2405.15216","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-automatic-speech-recognition-1","title":"Contextualized Automatic Speech Recognition with Dynamic Vocabulary","date":"2024-05-22","arxiv_id":"2405.13344","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-optimization-of-streaming-and-non","title":"Joint Optimization of Streaming and Non-Streaming Automatic Speech Recognition with Multi-Decoder and Knowledge Distillation","date":"2024-05-22","arxiv_id":"2405.13514","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-don-t-understand-me-comparing-asr-results","title":"You don't understand me!: Comparing ASR results for L1 and L2 speakers of Swedish","date":"2024-05-22","arxiv_id":"2405.13379","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairlens-assessing-fairness-in-law","title":"FairLENS: Assessing Fairness in Law Enforcement Speech Recognition","date":"2024-05-21","arxiv_id":"2405.13166","repositories_listed":0,"syntology":null},{"url":null,"slug":"listen-again-and-choose-the-right-answer-a","title":"Listen Again and Choose the Right Answer: A New Paradigm for Automatic Speech Recognition with Large Language Models","date":"2024-05-16","arxiv_id":"2405.10025","repositories_listed":0,"syntology":null},{"url":null,"slug":"continued-pretraining-for-domain-adaptation","title":"Continued Pretraining for Domain Adaptation of Wav2vec2.0 in Automatic Speech Recognition for Elementary Math Classroom Settings","date":"2024-05-15","arxiv_id":"2405.13018","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-evaluating-the-robustness-of","title":"Towards Evaluating the Robustness of Automatic Speech Recognition Systems via Audio Style Transfer","date":"2024-05-15","arxiv_id":"2405.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"sonos-voice-control-bias-assessment-dataset-a","title":"Sonos Voice Control Bias Assessment Dataset: A Methodology for Demographic Bias Assessment in Voice Assistants","date":"2024-05-14","arxiv_id":"2405.19342","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechverse-a-large-scale-generalizable-audio","title":"SpeechVerse: A Large-scale Generalizable Audio Language Model","date":"2024-05-14","arxiv_id":"2405.08295","repositories_listed":0,"syntology":null},{"url":null,"slug":"lost-in-transcription-identifying-and","title":"Lost in Transcription: Identifying and Quantifying the Accuracy Biases of Automatic Speech Recognition Systems Against Disfluent Speech","date":"2024-05-10","arxiv_id":"2405.06150","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmger-multi-modal-and-multi-granularity","title":"MMGER: Multi-modal and Multi-granularity Generative Error Correction with LLM for Joint Accent and Speech Recognition","date":"2024-05-06","arxiv_id":"2405.03152","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-x-vectors-and-bayesian-batch-active","title":"Combining X-Vectors and Bayesian Batch Active Learning: Two-Stage Active Learning Pipeline for Speech Recognition","date":"2024-05-03","arxiv_id":"2406.02566","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-compression-of-multitask","title":"Efficient Compression of Multitask Multilingual Speech Models","date":"2024-05-02","arxiv_id":"2405.00966","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-membership-inference-in-asr-model","title":"Improving Membership Inference in ASR Model Auditing with Perturbed Loss Features","date":"2024-05-02","arxiv_id":"2405.01207","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-to-sequence-models-in-peer-to-peer","title":"Sequence-to-sequence models in peer-to-peer learning: A practical application","date":"2024-05-02","arxiv_id":"2406.02565","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-whisper-understand-swiss-german-an","title":"Does Whisper understand Swiss German? An automatic, qualitative, and human evaluation","date":"2024-04-30","arxiv_id":"2404.19310","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-system","title":"Automatic Speech Recognition System-Independent Word Error Rate Estimation","date":"2024-04-25","arxiv_id":"2404.16743","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-acoustic-models-for-automatic","title":"Developing Acoustic Models for Automatic Speech Recognition in Swedish","date":"2024-04-25","arxiv_id":"2404.16547","repositories_listed":0,"syntology":null},{"url":null,"slug":"u2-moe-scaling-4-7x-parameters-with-minimal","title":"U2++ MoE: Scaling 4.7x parameters with minimal impact on RTF","date":"2024-04-25","arxiv_id":"2404.16407","repositories_listed":0,"syntology":null},{"url":null,"slug":"gated-low-rank-adaptation-for-personalized","title":"Gated Low-rank Adaptation for personalized Code-Switching Automatic Speech Recognition on the low-spec devices","date":"2024-04-24","arxiv_id":"2406.02562","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-walls-pioneering-automatic-speech","title":"Breaking Walls: Pioneering Automatic Speech Recognition for Central Kurdish: End-to-End Transformer Paradigm","date":"2024-04-23","arxiv_id":"2406.02561","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-processing-distortions","title":"Rethinking Processing Distortions: Disentangling the Impact of Speech Enhancement Errors on Speech Recognition Performance","date":"2024-04-23","arxiv_id":"2404.14860","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-infusion-of-self-supervised","title":"Efficient infusion of self-supervised representations in Automatic Speech Recognition","date":"2024-04-19","arxiv_id":"2404.12628","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-neural-networks-to-recognize","title":"Artificial Neural Networks to Recognize Speakers Division from Continuous Bengali Speech","date":"2024-04-18","arxiv_id":"2404.15168","repositories_listed":0,"syntology":null},{"url":null,"slug":"anatomy-of-industrial-scale-multilingual-asr","title":"Anatomy of Industrial Scale Multilingual ASR","date":"2024-04-15","arxiv_id":"2404.09841","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilience-of-large-language-models-for-noisy","title":"Resilience of Large Language Models for Noisy Instructions","date":"2024-04-15","arxiv_id":"2404.09754","repositories_listed":0,"syntology":null},{"url":"/paper/asr-advancements-for-indigenous-languages","slug":"asr-advancements-for-indigenous-languages","title":"Automatic Speech Recognition Advancements for Indigenous Languages of the Americas","date":"2024-04-12","arxiv_id":"2404.08368","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asr-advancements-for-indigenous-languages#ran","syntology_url":"https://syntology.ai/paper/2404.08368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08368"}},"official":null}},{"url":null,"slug":"comparing-apples-to-oranges-llm-powered","title":"Comparing Apples to Oranges: LLM-powered Multimodal Intention Prediction in an Object Categorization Task","date":"2024-04-12","arxiv_id":"2404.08424","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-automated-speaking-assessment","title":"An Effective Automated Speaking Assessment Approach to Mitigating Data Scarcity and Imbalanced Distribution","date":"2024-04-11","arxiv_id":"2404.07575","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformer-1-robust-asr-via-large-scale","title":"Conformer-1: Robust ASR via Large-Scale Semisupervised Bootstrapping","date":"2024-04-10","arxiv_id":"2404.07341","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-x-lance-technical-report-for-interspeech","title":"The X-LANCE Technical Report for Interspeech 2024 Speech Processing Using Discrete Speech Unit Challenge","date":"2024-04-09","arxiv_id":"2404.06079","repositories_listed":0,"syntology":null},{"url":null,"slug":"transducers-with-pronunciation-aware","title":"Transducers with Pronunciation-aware Embeddings for Automatic Speech Recognition","date":"2024-04-04","arxiv_id":"2404.04295","repositories_listed":0,"syntology":null},{"url":null,"slug":"mai-ho-omauna-i-ka-ai-language-models-improve","title":"Mai Ho'omāuna i ka 'Ai: Language Models Improve Automatic Speech Recognition in Hawaiian","date":"2024-04-03","arxiv_id":"2404.03073","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-masking-attacks-and-defenses-for","title":"Noise Masking Attacks and Defenses for Pretrained Speech Models","date":"2024-04-02","arxiv_id":"2404.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-from-whisper-for","title":"Transfer Learning from Whisper for Microscopic Intelligibility Prediction","date":"2024-04-02","arxiv_id":"2404.01737","repositories_listed":0,"syntology":null},{"url":null,"slug":"houston-we-have-a-divergence-a-subgroup","title":"Houston we have a Divergence: A Subgroup Performance Analysis of ASR Models","date":"2024-03-31","arxiv_id":"2404.07226","repositories_listed":0,"syntology":null},{"url":null,"slug":"lv-ctc-non-autoregressive-asr-with-ctc-and","title":"LV-CTC: Non-autoregressive ASR with CTC and latent variable models","date":"2024-03-28","arxiv_id":"2403.19207","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-multi-modal-pre-training-for","title":"Multi-Stage Multi-Modal Pre-Training for Automatic Speech Recognition","date":"2024-03-28","arxiv_id":"2403.19822","repositories_listed":0,"syntology":null},{"url":null,"slug":"zaebuc-spoken-a-multilingual-multidialectal","title":"ZAEBUC-Spoken: A Multilingual Multidialectal Arabic-English Speech Corpus","date":"2024-03-27","arxiv_id":"2403.18182","repositories_listed":0,"syntology":null},{"url":null,"slug":"dancer-entity-description-augmented-named","title":"DANCER: Entity Description Augmented Named Entity Corrector for Automatic Speech Recognition","date":"2024-03-26","arxiv_id":"2403.17645","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-biomedical-entities-from-noisy","title":"Extracting Biomedical Entities from Noisy Audio Transcripts","date":"2024-03-26","arxiv_id":"2403.17363","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-recurrent-adapters-for-efficient","title":"Hierarchical Recurrent Adapters for Efficient Multi-Task Adaptation of Large Speech Models","date":"2024-03-25","arxiv_id":"2403.19709","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-approach-to-device-directed","title":"A Multimodal Approach to Device-Directed Speech Detection with Large Language Models","date":"2024-03-21","arxiv_id":"2403.14438","repositories_listed":0,"syntology":null},{"url":null,"slug":"banglanum-a-public-dataset-for-bengali-digit","title":"BanglaNum -- A Public Dataset for Bengali Digit Recognition from Speech","date":"2024-03-20","arxiv_id":"2403.13465","repositories_listed":0,"syntology":null},{"url":null,"slug":"isometric-neural-machine-translation-using","title":"Isometric Neural Machine Translation using Phoneme Count Ratio Reward-based Reinforcement Learning","date":"2024-03-20","arxiv_id":"2403.15469","repositories_listed":0,"syntology":null},{"url":null,"slug":"adamer-ctc-connectionist-temporal","title":"AdaMER-CTC: Connectionist Temporal Classification with Adaptive Maximum Entropy Regularization for Automatic Speech Recognition","date":"2024-03-18","arxiv_id":"2403.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-artificial-intelligence-algorithms","title":"Artificial Intelligence for Cochlear Implants: Review of Strategies, Challenges, and Perspectives","date":"2024-03-17","arxiv_id":"2403.15442","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-asr-for-the","title":"Automatic Speech Recognition (ASR) for the Diagnosis of pronunciation of Speech Sound Disorders in Korean children","date":"2024-03-13","arxiv_id":"2403.08187","repositories_listed":0,"syntology":null},{"url":null,"slug":"skipformer-a-skip-and-recover-strategy-for","title":"Skipformer: A Skip-and-Recover Strategy for Efficient Speech Recognition","date":"2024-03-13","arxiv_id":"2403.08258","repositories_listed":0,"syntology":null},{"url":null,"slug":"gujarati-english-code-switching-speech","title":"Gujarati-English Code-Switching Speech Recognition using ensemble prediction of spoken language","date":"2024-03-12","arxiv_id":"2403.08011","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-evaluation-of-a-code-switched-sepedi","title":"The evaluation of a code-switched Sepedi-English automatic speech recognition system","date":"2024-03-11","arxiv_id":"2403.07947","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-speech-to-languages-to-enhance-code","title":"Aligning Speech to Languages to Enhance Code-switching Speech Recognition","date":"2024-03-09","arxiv_id":"2403.05887","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-benchmark-for-evaluating-automatic","title":"A New Benchmark for Evaluating Automatic Speech Recognition in the Arabic Call Domain","date":"2024-03-07","arxiv_id":"2403.04280","repositories_listed":0,"syntology":null},{"url":null,"slug":"classist-tools-social-class-correlates-with","title":"Classist Tools: Social Class Correlates with Performance in NLP","date":"2024-03-07","arxiv_id":"2403.04445","repositories_listed":0,"syntology":null},{"url":null,"slug":"jep-kd-joint-embedding-predictive","title":"JEP-KD: Joint-Embedding Predictive Architecture Based Knowledge Distillation for Visual Speech Recognition","date":"2024-03-04","arxiv_id":"2403.18843","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-has-lebenchmark-learnt-about-french","title":"What has LeBenchmark Learnt about French Syntax?","date":"2024-03-04","arxiv_id":"2403.02173","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-wav2vec2-embeddings-for-on","title":"A Closer Look at Wav2Vec2 Embeddings for On-Device Single-Channel Speech Enhancement","date":"2024-03-03","arxiv_id":"2403.01369","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-using-advanced","title":"Automatic Speech Recognition using Advanced Deep Learning Approaches: A survey","date":"2024-03-02","arxiv_id":"2403.01255","repositories_listed":0,"syntology":null},{"url":null,"slug":"post-decoder-biasing-for-end-to-end-speech","title":"Post-decoder Biasing for End-to-End Speech Recognition of Multi-turn Medical Interview","date":"2024-03-01","arxiv_id":"2403.00370","repositories_listed":0,"syntology":null},{"url":null,"slug":"inappropriate-pause-detection-in-dysarthric","title":"Inappropriate Pause Detection In Dysarthric Speech Using Large-Scale Speech Recognition","date":"2024-02-29","arxiv_id":"2402.18923","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-the-information-encoded-in-neural","title":"Probing the Information Encoded in Neural-based Acoustic Models of Automatic Speech Recognition Systems","date":"2024-02-29","arxiv_id":"2402.19443","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-adapter-for-noise-robust","title":"Exploration of Adapter for Noise Robust Automatic Speech Recognition","date":"2024-02-28","arxiv_id":"2402.18275","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-mixture-of-experts-approach-for","title":"An Effective Mixture-Of-Experts Approach For Code-Switching Speech Recognition Leveraging Encoder Disentanglement","date":"2024-02-27","arxiv_id":"2402.17189","repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-encoder-output-frame-rate-reduction","title":"Extreme Encoder Output Frame Rate Reduction: Improving Computational Latencies of Large End-to-End Models","date":"2024-02-27","arxiv_id":"2402.17184","repositories_listed":0,"syntology":null},{"url":null,"slug":"mel-fullsubnet-mel-spectrogram-enhancement","title":"Mel-FullSubNet: Mel-Spectrogram Enhancement for Improving Both Speech Quality and ASR","date":"2024-02-21","arxiv_id":"2402.13511","repositories_listed":0,"syntology":null},{"url":null,"slug":"ain-t-misbehavin-using-llms-to-generate","title":"Ain't Misbehavin' -- Using LLMs to Generate Expressive Robot Behavior in Conversations with the Tabletop Robot Haru","date":"2024-02-18","arxiv_id":"2402.11571","repositories_listed":0,"syntology":null},{"url":null,"slug":"unienc-cassnat-an-encoder-only-non","title":"UniEnc-CASSNAT: An Encoder-only Non-autoregressive ASR for Speech SSL Models","date":"2024-02-14","arxiv_id":"2402.08898","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-balancing-act-unmasking-and-alleviating","title":"The Balancing Act: Unmasking and Alleviating ASR Biases in Portuguese","date":"2024-02-12","arxiv_id":"2402.07513","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-sound-of-healthcare-improving-medical","title":"The Sound of Healthcare: Improving Medical Transcription ASR Accuracy with Large Language Models","date":"2024-02-12","arxiv_id":"2402.07658","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-consistent-context-aware-conformer","title":"Self-consistent context aware conformer transducer for speech recognition","date":"2024-02-09","arxiv_id":"2402.06592","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-unsupervised-domain-adaptation","title":"Progressive unsupervised domain adaptation for ASR using ensemble models and multi-stage training","date":"2024-02-07","arxiv_id":"2402.04805","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-study-of-the-current-state-of","title":"A Comprehensive Study of the Current State-of-the-Art in Nepali Automatic Speech Recognition Systems","date":"2024-02-05","arxiv_id":"2402.03050","repositories_listed":0,"syntology":null},{"url":null,"slug":"resolving-transcription-ambiguity-in-spanish","title":"Resolving Transcription Ambiguity in Spanish: A Hybrid Acoustic-Lexical System for Punctuation Restoration","date":"2024-02-05","arxiv_id":"2402.03519","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-positive-transfer-for-improved-low","title":"Predicting positive transfer for improved low-resource speech recognition using acoustic pseudo-tokens","date":"2024-02-03","arxiv_id":"2402.02302","repositories_listed":0,"syntology":null},{"url":null,"slug":"accentfold-a-journey-through-african-accents","title":"AccentFold: A Journey through African Accents for Zero-Shot ASR Adaptation to Target Accents","date":"2024-02-02","arxiv_id":"2402.01152","repositories_listed":0,"syntology":null},{"url":null,"slug":"digits-micro-model-for-accurate-and-secure","title":"Digits micro-model for accurate and secure transactions","date":"2024-02-02","arxiv_id":"2402.01931","repositories_listed":0,"syntology":null},{"url":null,"slug":"whispering-in-norwegian-navigating","title":"Whispering in Norwegian: Navigating Orthographic and Dialectic Challenges","date":"2024-02-02","arxiv_id":"2402.01917","repositories_listed":0,"syntology":null},{"url":null,"slug":"byte-pair-encoding-is-all-you-need-for","title":"Byte Pair Encoding Is All You Need For Automatic Bengali Speech Recognition","date":"2024-01-28","arxiv_id":"2401.15532","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-practical-automatic-speech-recognition","title":"Toward Practical Automatic Speech Recognition and Post-Processing: a Call for Explainable Error Benchmark Guideline","date":"2024-01-26","arxiv_id":"2401.14625","repositories_listed":0,"syntology":null},{"url":null,"slug":"mf-aed-aec-speech-emotion-recognition-by","title":"MF-AED-AEC: Speech Emotion Recognition by Leveraging Multimodal Fusion, Asr Error Detection, and Asr Error Correction","date":"2024-01-24","arxiv_id":"2401.13260","repositories_listed":0,"syntology":null},{"url":null,"slug":"locality-enhanced-dynamic-biasing-and","title":"Locality enhanced dynamic biasing and sampling strategies for contextual ASR","date":"2024-01-23","arxiv_id":"2401.13146","repositories_listed":0,"syntology":null},{"url":null,"slug":"consistency-based-unsupervised-self-training","title":"Consistency Based Unsupervised Self-training For ASR Personalisation","date":"2024-01-22","arxiv_id":"2401.12085","repositories_listed":0,"syntology":null},{"url":null,"slug":"keep-decoding-parallel-with-effective","title":"Keep Decoding Parallel with Effective Knowledge Distillation from Language Models to End-to-end Speech Recognisers","date":"2024-01-22","arxiv_id":"2401.11700","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-large-language-model-for-end-to-end","title":"Using Large Language Model for End-to-End Chinese ASR and NER","date":"2024-01-21","arxiv_id":"2401.11382","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-automatic-speech-recognition","title":"Contextualized Automatic Speech Recognition with Attention-Based Bias Phrase Boosted Beam Search","date":"2024-01-19","arxiv_id":"2401.10449","repositories_listed":0,"syntology":null},{"url":null,"slug":"agadir-towards-array-geometry-agnostic","title":"AGADIR: Towards Array-Geometry Agnostic Directional Speech Recognition","date":"2024-01-18","arxiv_id":"2401.10411","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-personalized-1","title":"Communication-Efficient Personalized Federated Learning for Speech-to-Text Tasks","date":"2024-01-18","arxiv_id":"2401.10070","repositories_listed":0,"syntology":null},{"url":null,"slug":"slideavsr-a-dataset-of-paper-explanation","title":"SlideAVSR: A Dataset of Paper Explanation Videos for Audio-Visual Speech Recognition","date":"2024-01-18","arxiv_id":"2401.09759","repositories_listed":0,"syntology":null}],"record_sha256":"4af237db7cd0bd62b7310084341a7bc07f49de56a4344dc8b3d279174df1ced8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}