{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/21","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":21,"pages_in_order":65,"rows_per_page":100,"rows":[2001,2100],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/20","next":"/task/speech-recognition/papers/22","papers":[{"url":null,"slug":"a-parameter-efficient-language-extension","title":"A Parameter-efficient Language Extension Framework for Multilingual ASR","date":"2024-06-10","arxiv_id":"2406.06329","repositories_listed":0,"syntology":null},{"url":null,"slug":"astra-aligning-speech-and-text","title":"ASTRA: Aligning Speech and Text Representations for Asr without Sampling","date":"2024-06-10","arxiv_id":"2406.06664","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-query-generation-using-large","title":"Synthetic Query Generation using Large Language Models for Virtual Assistants","date":"2024-06-10","arxiv_id":"2406.06729","repositories_listed":0,"syntology":null},{"url":null,"slug":"ms-hubert-mitigating-pre-training-and","title":"MS-HuBERT: Mitigating Pre-training and Inference Mismatch in Masked Language Modelling methods for learning Speech Representations","date":"2024-06-09","arxiv_id":"2406.05661","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-multi-stuttered-speech","title":"Optimizing Multi-Stuttered Speech Classification: Leveraging Whisper's Encoder for Efficient Parameter Reduction in Automated Assessment","date":"2024-06-09","arxiv_id":"2406.05784","repositories_listed":0,"syntology":null},{"url":null,"slug":"lora-whisper-parameter-efficient-and","title":"LoRA-Whisper: Parameter-Efficient and Extensible Multilingual ASR","date":"2024-06-07","arxiv_id":"2406.06619","repositories_listed":0,"syntology":null},{"url":null,"slug":"pitch-aware-rnn-t-for-mandarin-chinese","title":"Pitch-Aware RNN-T for Mandarin Chinese Mispronunciation Detection and Diagnosis","date":"2024-06-07","arxiv_id":"2406.04595","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-multichannel-speech-enhancement-for","title":"Flexible Multichannel Speech Enhancement for Noise-Robust Frontend","date":"2024-06-06","arxiv_id":"2406.04552","repositories_listed":0,"syntology":null},{"url":null,"slug":"helsinki-speech-challenge-2024","title":"Helsinki Speech Challenge 2024","date":"2024-06-06","arxiv_id":"2406.04123","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypernetworks-for-personalizing-asr-to","title":"Hypernetworks for Personalizing ASR to Atypical Speech","date":"2024-06-06","arxiv_id":"2406.04240","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-zero-shot-chinese-english-code","title":"Improving Zero-Shot Chinese-English Code-Switching ASR with kNN-CTC and Gated Monolingual Datastores","date":"2024-06-06","arxiv_id":"2406.03814","repositories_listed":0,"syntology":null},{"url":null,"slug":"speed-of-light-exact-greedy-decoding-for-rnn","title":"Speed of Light Exact Greedy Decoding for RNN-T Speech Recognition Models on GPU","date":"2024-06-06","arxiv_id":"2406.03791","repositories_listed":0,"syntology":null},{"url":null,"slug":"4d-asr-joint-beam-search-integrating-ctc","title":"Joint Beam Search Integrating CTC, Attention, and Transducer Decoders","date":"2024-06-05","arxiv_id":"2406.02950","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-ctc-based-speech-recognition-with","title":"Enhancing CTC-based speech recognition with diverse modeling units","date":"2024-06-05","arxiv_id":"2406.03274","repositories_listed":0,"syntology":null},{"url":null,"slug":"syn2real-leveraging-task-arithmetic-for","title":"Task Arithmetic can Mitigate Synthetic-to-Real Gap in Automatic Speech Recognition","date":"2024-06-05","arxiv_id":"2406.02925","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-injection-for-neural-contextual-biasing","title":"Text Injection for Neural Contextual Biasing","date":"2024-06-05","arxiv_id":"2406.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-multimodal-transformers-with-a","title":"Discrete Multimodal Transformers with a Pretrained Large Language Model for Mixed-Supervision Speech Processing","date":"2024-06-04","arxiv_id":"2406.06582","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiently-train-asr-models-that-memorize","title":"Efficiently Train ASR Models that Memorize Less and Perform Better with Per-core Clipping","date":"2024-06-04","arxiv_id":"2406.02004","repositories_listed":0,"syntology":null},{"url":null,"slug":"keyword-guided-adaptation-of-automatic-speech","title":"Keyword-Guided Adaptation of Automatic Speech Recognition","date":"2024-06-04","arxiv_id":"2406.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"compute-efficient-medical-image","title":"Compute-Efficient Medical Image Classification with Softmax-Free Transformers and Sequence Normalization","date":"2024-06-03","arxiv_id":"2406.01314","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-asr-for-low-resource-languages-a","title":"Enabling ASR for Low-Resource Languages: A Comprehensive Dataset Creation Approach","date":"2024-06-03","arxiv_id":"2406.01446","repositories_listed":0,"syntology":null},{"url":null,"slug":"yodas-youtube-oriented-dataset-for-audio-and","title":"YODAS: Youtube-Oriented Dataset for Audio and Speech","date":"2024-06-02","arxiv_id":"2406.00899","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2prompt-end-to-end-speech-prompt","title":"Wav2Prompt: End-to-End Speech Prompt Generation and Tuning For LLM in Zero and Few-shot Learning","date":"2024-06-01","arxiv_id":"2406.00522","repositories_listed":0,"syntology":null},{"url":null,"slug":"zipper-a-multi-tower-decoder-architecture-for","title":"Zipper: A Multi-Tower Decoder Architecture for Fusing Modalities","date":"2024-05-29","arxiv_id":"2405.18669","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-conversation-with-embedded-speech","title":"Augmented Conversation with Embedded Speech-Driven On-the-Fly Referencing in AR","date":"2024-05-28","arxiv_id":"2405.18537","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-clinical-documentation-harnessing","title":"Intelligent Clinical Documentation: Harnessing Generative AI for Patient-Centric Clinical Note Generation","date":"2024-05-28","arxiv_id":"2405.18346","repositories_listed":0,"syntology":null},{"url":null,"slug":"nuts-nars-and-speech","title":"NUTS, NARS, and Speech","date":"2024-05-28","arxiv_id":"2405.17874","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-lm-pushing-the-limits-of-error","title":"Denoising LM: Pushing the Limits of Error Correction Models for Speech Recognition","date":"2024-05-24","arxiv_id":"2405.15216","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-automatic-speech-recognition-1","title":"Contextualized Automatic Speech Recognition with Dynamic Vocabulary","date":"2024-05-22","arxiv_id":"2405.13344","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-optimization-of-streaming-and-non","title":"Joint Optimization of Streaming and Non-Streaming Automatic Speech Recognition with Multi-Decoder and Knowledge Distillation","date":"2024-05-22","arxiv_id":"2405.13514","repositories_listed":0,"syntology":null},{"url":null,"slug":"st-gait-leveraging-spatio-temporal","title":"ST-Gait++: Leveraging spatio-temporal convolutions for gait-based emotion recognition on videos","date":"2024-05-22","arxiv_id":"2405.13903","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-don-t-understand-me-comparing-asr-results","title":"You don't understand me!: Comparing ASR results for L1 and L2 speakers of Swedish","date":"2024-05-22","arxiv_id":"2405.13379","repositories_listed":0,"syntology":null},{"url":null,"slug":"could-a-computer-architect-understand-our","title":"Could a Computer Architect Understand our Brain?","date":"2024-05-21","arxiv_id":"2405.12815","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairlens-assessing-fairness-in-law","title":"FairLENS: Assessing Fairness in Law Enforcement Speech Recognition","date":"2024-05-21","arxiv_id":"2405.13166","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-real-time-accent","title":"Non-autoregressive real-time Accent Conversion model with voice cloning","date":"2024-05-21","arxiv_id":"2405.13162","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-sign-language-recognition-with-1","title":"Continuous Sign Language Recognition with Adapted Conformer via Unsupervised Pretraining","date":"2024-05-20","arxiv_id":"2405.12018","repositories_listed":0,"syntology":null},{"url":null,"slug":"listen-again-and-choose-the-right-answer-a","title":"Listen Again and Choose the Right Answer: A New Paradigm for Automatic Speech Recognition with Large Language Models","date":"2024-05-16","arxiv_id":"2405.10025","repositories_listed":0,"syntology":null},{"url":null,"slug":"continued-pretraining-for-domain-adaptation","title":"Continued Pretraining for Domain Adaptation of Wav2vec2.0 in Automatic Speech Recognition for Elementary Math Classroom Settings","date":"2024-05-15","arxiv_id":"2405.13018","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-evaluating-the-robustness-of","title":"Towards Evaluating the Robustness of Automatic Speech Recognition Systems via Audio Style Transfer","date":"2024-05-15","arxiv_id":"2405.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-autoencoder-behavior-in","title":"Investigating the 'Autoencoder Behavior' in Speech Self-Supervised Models: a focus on HuBERT's Pretraining","date":"2024-05-14","arxiv_id":"2405.08402","repositories_listed":0,"syntology":null},{"url":null,"slug":"sonos-voice-control-bias-assessment-dataset-a","title":"Sonos Voice Control Bias Assessment Dataset: A Methodology for Demographic Bias Assessment in Voice Assistants","date":"2024-05-14","arxiv_id":"2405.19342","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechverse-a-large-scale-generalizable-audio","title":"SpeechVerse: A Large-scale Generalizable Audio Language Model","date":"2024-05-14","arxiv_id":"2405.08295","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-education-a-survey-1","title":"Large Language Models for Education: A Survey","date":"2024-05-12","arxiv_id":"2405.13001","repositories_listed":0,"syntology":null},{"url":null,"slug":"dp-dylora-fine-tuning-transformer-based","title":"DP-DyLoRA: Fine-Tuning Transformer-Based Models On-Device under Differentially Private Federated Learning using Dynamic Low-Rank Adaptation","date":"2024-05-10","arxiv_id":"2405.06368","repositories_listed":0,"syntology":null},{"url":null,"slug":"lost-in-transcription-identifying-and","title":"Lost in Transcription: Identifying and Quantifying the Accuracy Biases of Automatic Speech Recognition Systems Against Disfluent Speech","date":"2024-05-10","arxiv_id":"2405.06150","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmger-multi-modal-and-multi-granularity","title":"MMGER: Multi-modal and Multi-granularity Generative Error Correction with LLM for Joint Accent and Speech Recognition","date":"2024-05-06","arxiv_id":"2405.03152","repositories_listed":0,"syntology":null},{"url":null,"slug":"whispy-adapting-stt-whisper-models-to-real","title":"Whispy: Adapting STT Whisper Models to Real-Time Environments","date":"2024-05-06","arxiv_id":"2405.03484","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-x-vectors-and-bayesian-batch-active","title":"Combining X-Vectors and Bayesian Batch Active Learning: Two-Stage Active Learning Pipeline for Speech Recognition","date":"2024-05-03","arxiv_id":"2406.02566","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-models-in-speech-recognition","title":"Deep Learning Models in Speech Recognition: Measuring GPU Energy Consumption, Impact of Noise and Model Quantization for Edge Deployment","date":"2024-05-02","arxiv_id":"2405.01004","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-compression-of-multitask","title":"Efficient Compression of Multitask Multilingual Speech Models","date":"2024-05-02","arxiv_id":"2405.00966","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-membership-inference-in-asr-model","title":"Improving Membership Inference in ASR Model Auditing with Perturbed Loss Features","date":"2024-05-02","arxiv_id":"2405.01207","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-speech-recognition-and-dialect","title":"Low-resource speech recognition and dialect identification of Irish in a multi-task framework","date":"2024-05-02","arxiv_id":"2405.01293","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-to-sequence-models-in-peer-to-peer","title":"Sequence-to-sequence models in peer-to-peer learning: A practical application","date":"2024-05-02","arxiv_id":"2406.02565","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-with-task-adaptation-pre","title":"Active Learning with Task Adaptation Pre-training for Speech Emotion Recognition","date":"2024-05-01","arxiv_id":"2405.00307","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-sample-specific-encoder","title":"Efficient Sample-Specific Encoder Perturbations","date":"2024-05-01","arxiv_id":"2405.01601","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-whisper-understand-swiss-german-an","title":"Does Whisper understand Swiss German? An automatic, qualitative, and human evaluation","date":"2024-04-30","arxiv_id":"2404.19310","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cost-minimization-approach-to-fix-the","title":"A cost minimization approach to fix the vocabulary size in a tokenizer for an End-to-End ASR system","date":"2024-04-29","arxiv_id":"2406.02563","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-dog-bark-decoding-leveraging-human","title":"Towards Dog Bark Decoding: Leveraging Human Speech Processing for Automated Bark Classification","date":"2024-04-29","arxiv_id":"2404.18739","repositories_listed":0,"syntology":null},{"url":null,"slug":"child-speech-recognition-in-human-robot","title":"Child Speech Recognition in Human-Robot Interaction: Problem Solved?","date":"2024-04-26","arxiv_id":"2404.17394","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-system","title":"Automatic Speech Recognition System-Independent Word Error Rate Estimation","date":"2024-04-25","arxiv_id":"2404.16743","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-acoustic-models-for-automatic","title":"Developing Acoustic Models for Automatic Speech Recognition in Swedish","date":"2024-04-25","arxiv_id":"2404.16547","repositories_listed":0,"syntology":null},{"url":null,"slug":"u2-moe-scaling-4-7x-parameters-with-minimal","title":"U2++ MoE: Scaling 4.7x parameters with minimal impact on RTF","date":"2024-04-25","arxiv_id":"2404.16407","repositories_listed":0,"syntology":null},{"url":null,"slug":"gated-low-rank-adaptation-for-personalized","title":"Gated Low-rank Adaptation for personalized Code-Switching Automatic Speech Recognition on the low-spec devices","date":"2024-04-24","arxiv_id":"2406.02562","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-walls-pioneering-automatic-speech","title":"Breaking Walls: Pioneering Automatic Speech Recognition for Central Kurdish: End-to-End Transformer Paradigm","date":"2024-04-23","arxiv_id":"2406.02561","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-processing-distortions","title":"Rethinking Processing Distortions: Disentangling the Impact of Speech Enhancement Errors on Speech Recognition Performance","date":"2024-04-23","arxiv_id":"2404.14860","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-infusion-of-self-supervised","title":"Efficient infusion of self-supervised representations in Automatic Speech Recognition","date":"2024-04-19","arxiv_id":"2404.12628","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn2talk-3d-talking-face-learns-from-2d","title":"Learn2Talk: 3D Talking Face Learns from 2D Talking Face","date":"2024-04-19","arxiv_id":"2404.12888","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-neural-networks-to-recognize","title":"Artificial Neural Networks to Recognize Speakers Division from Continuous Bengali Speech","date":"2024-04-18","arxiv_id":"2404.15168","repositories_listed":0,"syntology":null},{"url":null,"slug":"anatomy-of-industrial-scale-multilingual-asr","title":"Anatomy of Industrial Scale Multilingual ASR","date":"2024-04-15","arxiv_id":"2404.09841","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilience-of-large-language-models-for-noisy","title":"Resilience of Large Language Models for Noisy Instructions","date":"2024-04-15","arxiv_id":"2404.09754","repositories_listed":0,"syntology":null},{"url":"/paper/asr-advancements-for-indigenous-languages","slug":"asr-advancements-for-indigenous-languages","title":"Automatic Speech Recognition Advancements for Indigenous Languages of the Americas","date":"2024-04-12","arxiv_id":"2404.08368","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asr-advancements-for-indigenous-languages#ran","syntology_url":"https://syntology.ai/paper/2404.08368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08368"}},"official":null}},{"url":null,"slug":"comparing-apples-to-oranges-llm-powered","title":"Comparing Apples to Oranges: LLM-powered Multimodal Intention Prediction in an Object Categorization Task","date":"2024-04-12","arxiv_id":"2404.08424","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-automated-speaking-assessment","title":"An Effective Automated Speaking Assessment Approach to Mitigating Data Scarcity and Imbalanced Distribution","date":"2024-04-11","arxiv_id":"2404.07575","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-inclusive-review-on-deep-learning","title":"An inclusive review on deep learning techniques and their scope in handwriting recognition","date":"2024-04-10","arxiv_id":"2404.08011","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformer-1-robust-asr-via-large-scale","title":"Conformer-1: Robust ASR via Large-Scale Semisupervised Bootstrapping","date":"2024-04-10","arxiv_id":"2404.07341","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-x-lance-technical-report-for-interspeech","title":"The X-LANCE Technical Report for Interspeech 2024 Speech Processing Using Discrete Speech Unit Challenge","date":"2024-04-09","arxiv_id":"2404.06079","repositories_listed":0,"syntology":null},{"url":null,"slug":"transducers-with-pronunciation-aware","title":"Transducers with Pronunciation-aware Embeddings for Automatic Speech Recognition","date":"2024-04-04","arxiv_id":"2404.04295","repositories_listed":0,"syntology":null},{"url":null,"slug":"mai-ho-omauna-i-ka-ai-language-models-improve","title":"Mai Ho'omāuna i ka 'Ai: Language Models Improve Automatic Speech Recognition in Hawaiian","date":"2024-04-03","arxiv_id":"2404.03073","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-masking-attacks-and-defenses-for","title":"Noise Masking Attacks and Defenses for Pretrained Speech Models","date":"2024-04-02","arxiv_id":"2404.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-from-whisper-for","title":"Transfer Learning from Whisper for Microscopic Intelligibility Prediction","date":"2024-04-02","arxiv_id":"2404.01737","repositories_listed":0,"syntology":null},{"url":null,"slug":"houston-we-have-a-divergence-a-subgroup","title":"Houston we have a Divergence: A Subgroup Performance Analysis of ASR Models","date":"2024-03-31","arxiv_id":"2404.07226","repositories_listed":0,"syntology":null},{"url":null,"slug":"lv-ctc-non-autoregressive-asr-with-ctc-and","title":"LV-CTC: Non-autoregressive ASR with CTC and latent variable models","date":"2024-03-28","arxiv_id":"2403.19207","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-multi-modal-pre-training-for","title":"Multi-Stage Multi-Modal Pre-Training for Automatic Speech Recognition","date":"2024-03-28","arxiv_id":"2403.19822","repositories_listed":0,"syntology":null},{"url":null,"slug":"zaebuc-spoken-a-multilingual-multidialectal","title":"ZAEBUC-Spoken: A Multilingual Multidialectal Arabic-English Speech Corpus","date":"2024-03-27","arxiv_id":"2403.18182","repositories_listed":0,"syntology":null},{"url":null,"slug":"dancer-entity-description-augmented-named","title":"DANCER: Entity Description Augmented Named Entity Corrector for Automatic Speech Recognition","date":"2024-03-26","arxiv_id":"2403.17645","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-biomedical-entities-from-noisy","title":"Extracting Biomedical Entities from Noisy Audio Transcripts","date":"2024-03-26","arxiv_id":"2403.17363","repositories_listed":0,"syntology":null},{"url":null,"slug":"grammatical-vs-spelling-error-correction-an","title":"Grammatical vs Spelling Error Correction: An Investigation into the Responsiveness of Transformer-based Language Models using BART and MarianMT","date":"2024-03-25","arxiv_id":"2403.16655","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-recurrent-adapters-for-efficient","title":"Hierarchical Recurrent Adapters for Efficient Multi-Task Adaptation of Large Speech Models","date":"2024-03-25","arxiv_id":"2403.19709","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-end-to-end-spoken-language","title":"Privacy-Preserving End-to-End Spoken Language Understanding","date":"2024-03-22","arxiv_id":"2403.15510","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-approach-to-device-directed","title":"A Multimodal Approach to Device-Directed Speech Detection with Large Language Models","date":"2024-03-21","arxiv_id":"2403.14438","repositories_listed":0,"syntology":null},{"url":"/paper/m-3-av-a-multimodal-multigenre-and","slug":"m-3-av-a-multimodal-multigenre-and","title":"M$^3$AV: A Multimodal, Multigenre, and Multipurpose Audio-Visual Academic Lecture Dataset","date":"2024-03-21","arxiv_id":"2403.14168","repositories_listed":0,"syntology":null},{"url":null,"slug":"xlavs-r-cross-lingual-audio-visual-speech","title":"XLAVS-R: Cross-Lingual Audio-Visual Speech Representation Learning for Noise-Robust Speech Perception","date":"2024-03-21","arxiv_id":"2403.14402","repositories_listed":0,"syntology":null},{"url":null,"slug":"banglanum-a-public-dataset-for-bengali-digit","title":"BanglaNum -- A Public Dataset for Bengali Digit Recognition from Speech","date":"2024-03-20","arxiv_id":"2403.13465","repositories_listed":0,"syntology":null},{"url":null,"slug":"isometric-neural-machine-translation-using","title":"Isometric Neural Machine Translation using Phoneme Count Ratio Reward-based Reinforcement Learning","date":"2024-03-20","arxiv_id":"2403.15469","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-access-nao-oan-a-ros2-based-software","title":"Open Access NAO (OAN): a ROS2-based software framework for HRI applications with the NAO robot","date":"2024-03-20","arxiv_id":"2403.13960","repositories_listed":0,"syntology":null},{"url":null,"slug":"adamer-ctc-connectionist-temporal","title":"AdaMER-CTC: Connectionist Temporal Classification with Adaptive Maximum Entropy Regularization for Automatic Speech Recognition","date":"2024-03-18","arxiv_id":"2403.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-artificial-intelligence-algorithms","title":"Artificial Intelligence for Cochlear Implants: Review of Strategies, Challenges, and Perspectives","date":"2024-03-17","arxiv_id":"2403.15442","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-based-models-with-applications-to","title":"Energy-Based Models with Applications to Speech and Language Processing","date":"2024-03-16","arxiv_id":"2403.10961","repositories_listed":0,"syntology":null},{"url":null,"slug":"initial-decoding-with-minimally-augmented","title":"Initial Decoding with Minimally Augmented Language Model for Improved Lattice Rescoring in Low Resource ASR","date":"2024-03-16","arxiv_id":"2403.10937","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-networks-hear-you-loud-and-clear","title":"Hearing-Loss Compensation Using Deep Neural Networks: A Framework and Results From a Listening Test","date":"2024-03-15","arxiv_id":"2403.10420","repositories_listed":0,"syntology":null}],"record_sha256":"47b9b8c8868f56fce971d8c6389a615d88fd472b830660737ecd634cd166999c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}