{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/25","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":58,"rows_per_page":100,"rows":[2401,2500],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/24","next":"/task/speech-recognition-1/papers/26","papers":[{"url":null,"slug":"federated-self-learning-with-weak-supervision","title":"Federated Self-Learning with Weak Supervision for Speech Recognition","date":"2023-06-21","arxiv_id":"2306.12015","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-when-to-trust-which-teacher-for","title":"Learning When to Trust Which Teacher for Weakly Supervised ASR","date":"2023-06-21","arxiv_id":"2306.12012","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-encoder-for-joint-speech-separation","title":"Mixture Encoder for Joint Speech Separation and Recognition","date":"2023-06-21","arxiv_id":"2306.12173","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategies-in-transfer-learning-for-low","title":"Strategies in Transfer Learning for Low-Resource Speech Synthesis: Phone Mapping, Features Input, and Source Language Selection","date":"2023-06-21","arxiv_id":"2306.12040","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-pass-training-and-cross-information","title":"Multi-pass Training and Cross-information Fusion for Low-resource End-to-end Accented Speech Recognition","date":"2023-06-20","arxiv_id":"2306.11309","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillation-strategies-for-discriminative","title":"Distillation Strategies for Discriminative Speech Recognition Rescoring","date":"2023-06-15","arxiv_id":"2306.09452","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexical-speaker-error-correction-leveraging","title":"Lexical Speaker Error Correction: Leveraging Language Models for Speaker Diarization Error Correction","date":"2023-06-15","arxiv_id":"2306.09313","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobileasr-a-resource-aware-on-device","title":"MobileASR: A resource-aware on-device learning framework for user voice personalization applications on mobile phones","date":"2023-06-15","arxiv_id":"2306.09384","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-speaker-independent-visual-speech","title":"Automated Speaker Independent Visual Speech Recognition: A Comprehensive Survey","date":"2023-06-14","arxiv_id":"2306.08314","repositories_listed":0,"syntology":null},{"url":null,"slug":"em-network-oracle-guided-self-distillation","title":"EM-Network: Oracle Guided Self-distillation for Sequence Learning","date":"2023-06-14","arxiv_id":"2306.10058","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-normalization-for-fine-tuning-self","title":"Feature Normalization for Fine-tuning Self-Supervised Models in Speech Enhancement","date":"2023-06-14","arxiv_id":"2306.08406","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-code-switching-and-named-entity","title":"Improving Code-Switching and Named Entity Recognition in ASR with Speech Editing based Data Augmentation","date":"2023-06-14","arxiv_id":"2306.08588","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-cross-lingual-mappings-for-data","title":"Learning Cross-lingual Mappings for Data Augmentation to Improve Low-Resource Speech Recognition","date":"2023-06-14","arxiv_id":"2306.08577","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-on-an-improved-conformer-end-to-end","title":"Research on an improved Conformer end-to-end Speech Recognition Model with R-Drop Structure","date":"2023-06-14","arxiv_id":"2306.08329","repositories_listed":0,"syntology":null},{"url":null,"slug":"dctx-conformer-dynamic-context-carry-over-for","title":"DCTX-Conformer: Dynamic context carry-over for low latency unified streaming and non-streaming Conformer ASR","date":"2023-06-13","arxiv_id":"2306.08175","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-language-model-rescoring-on-long","title":"Large-scale Language Model Rescoring on Long-form Data","date":"2023-06-13","arxiv_id":"2306.08133","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-beamformer-exploiting-non","title":"Statistical Beamformer Exploiting Non-stationarity and Sparsity with Spatially Constrained ICA for Robust Speech Recognition","date":"2023-06-13","arxiv_id":"2306.07562","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-frequency-attention-alternative-to","title":"Multi-View Frequency-Attention Alternative to CNN Frontends for Automatic Speech Recognition","date":"2023-06-12","arxiv_id":"2306.06954","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-audio-textual-architecture-for-1","title":"Multimodal Audio-textual Architecture for Robust Spoken Language Understanding","date":"2023-06-12","arxiv_id":"2306.06819","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-n-gram-approximation-of-pre-trained","title":"On the N-gram Approximation of Pre-trained Language Models","date":"2023-06-12","arxiv_id":"2306.06892","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-dysarthric-speech","title":"Parameter-efficient Dysarthric Speech Recognition Using Adapter Fusion and Householder Transformation","date":"2023-06-12","arxiv_id":"2306.07090","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-experiencing-misrecognition-by","title":"Impact of Experiencing Misrecognition by Teachable Agents on Learning and Rapport","date":"2023-06-11","arxiv_id":"2306.07302","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-influence-in-multimodal-machine","title":"Modality Influence in Multimodal Machine Learning","date":"2023-06-10","arxiv_id":"2306.06476","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-can-an-accent-identifier-learn-probing","title":"What Can an Accent Identifier Learn? Probing Phonetic and Prosodic Information in a Wav2vec2-based Accent Identification Model","date":"2023-06-10","arxiv_id":"2306.06524","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-frame-level-classifier-for-word","title":"Improving Frame-level Classifier for Word Timings with Non-peaky CTC in End-to-End Automatic Speech Recognition","date":"2023-06-09","arxiv_id":"2306.07949","repositories_listed":0,"syntology":null},{"url":null,"slug":"record-deduplication-for-entity-distribution","title":"Record Deduplication for Entity Distribution Modeling in ASR Transcripts","date":"2023-06-09","arxiv_id":"2306.06246","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-language-model-integration-for","title":"Improving Language Model Integration for Neural Machine Translation","date":"2023-06-08","arxiv_id":"2306.05077","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-phrase-matching-for-dysarthric-speech","title":"Latent Phrase Matching for Dysarthric Speech","date":"2023-06-08","arxiv_id":"2306.05446","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-reliability-of-automatic","title":"A study on the impact of Self-Supervised Learning on automatic dysarthric speech assessment","date":"2023-06-07","arxiv_id":"2306.04337","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-asr-based-tutor-for-learning-to-read-how","title":"An ASR-Based Tutor for Learning to Read: How to Optimize Feedback to First Graders","date":"2023-06-07","arxiv_id":"2306.04190","repositories_listed":0,"syntology":null},{"url":null,"slug":"fooctts-generating-arabic-speech-with","title":"FOOCTTS: Generating Arabic Speech with Acoustic Environment for Football Commentator","date":"2023-06-07","arxiv_id":"2306.07936","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-aware-speech-representation-learning","title":"Label Aware Speech Representation Learning For Language Identification","date":"2023-06-07","arxiv_id":"2306.04374","repositories_listed":0,"syntology":null},{"url":null,"slug":"lenient-evaluation-of-japanese-speech","title":"Lenient Evaluation of Japanese Speech Recognition: Modeling Naturally Occurring Spelling Inconsistency","date":"2023-06-07","arxiv_id":"2306.04530","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-only-domain-adaptation-using-unified","title":"Text-only Domain Adaptation using Unified Speech-Text Representation in Transducer","date":"2023-06-07","arxiv_id":"2306.04076","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-from-pre-trained-language","title":"Transfer Learning from Pre-trained Language Models Improves End-to-End Speech Summarization","date":"2023-06-07","arxiv_id":"2306.04233","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-of-transformer-based-speech","title":"Transfer Learning of Transformer-based Speech Recognition Models from Czech to Slovak","date":"2023-06-07","arxiv_id":"2306.04399","repositories_listed":0,"syntology":null},{"url":null,"slug":"alzheimer-disease-classification-through-asr","title":"Alzheimer Disease Classification through ASR-based Transcriptions: Exploring the Impact of Punctuation and Pauses","date":"2023-06-06","arxiv_id":"2306.03443","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-assessment-of-oral-reading-accuracy","title":"Automatic Assessment of Oral Reading Accuracy for Reading Diagnostics","date":"2023-06-06","arxiv_id":"2306.03444","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-fairness-and-robustness-in-end-to","title":"Improving Fairness and Robustness in End-to-End Speech Recognition through unsupervised clustering","date":"2023-06-06","arxiv_id":"2306.06083","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-unlearning-a-survey","title":"Machine Unlearning: A Survey","date":"2023-06-06","arxiv_id":"2306.03558","repositories_listed":0,"syntology":null},{"url":null,"slug":"rescuespeech-a-german-corpus-for-speech","title":"RescueSpeech: A German Corpus for Speech Recognition in Search and Rescue Domain","date":"2023-06-06","arxiv_id":"2306.04054","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-l2-phonemes-using-articulatory","title":"Incorporating L2 Phonemes Using Articulatory Features for Robust Speech Recognition","date":"2023-06-05","arxiv_id":"2306.02534","repositories_listed":0,"syntology":null},{"url":null,"slug":"n-shot-benchmarking-of-whisper-on-diverse","title":"N-Shot Benchmarking of Whisper on Diverse Arabic Speech Recognition","date":"2023-06-05","arxiv_id":"2306.02902","repositories_listed":0,"syntology":null},{"url":null,"slug":"otf-optimal-transport-based-fusion-of","title":"OTF: Optimal Transport based Fusion of Supervised and Self-Supervised Learning Models for Automatic Speech Recognition","date":"2023-06-05","arxiv_id":"2306.02541","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-joint-target-and-non-target","title":"End-to-End Joint Target and Non-Target Speakers ASR","date":"2023-06-04","arxiv_id":"2306.02273","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-with-score","title":"Audio-Visual Speech Enhancement with Score-Based Generative Models","date":"2023-06-02","arxiv_id":"2306.01432","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-training-for-end-to-end-streaming","title":"Improved Training for End-to-End Streaming Automatic Speech Recognition Model with Punctuation","date":"2023-06-02","arxiv_id":"2306.01296","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-crowdsourcing-design-with-comparison","title":"On Crowdsourcing-design with Comparison Category Rating for Evaluating Speech Enhancement Algorithms","date":"2023-06-02","arxiv_id":"2306.01538","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-speech-to-confusion-network-speech","title":"Streaming Speech-to-Confusion Network Speech Recognition","date":"2023-06-02","arxiv_id":"2306.03778","repositories_listed":0,"syntology":null},{"url":null,"slug":"tensor-decomposition-for-minimization-of-e2e","title":"Tensor decomposition for minimization of E2E SLU model toward on-device processing","date":"2023-06-02","arxiv_id":"2306.01247","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptation-and-optimization-of-automatic","title":"Adaptation and Optimization of Automatic Speech Recognition (ASR) for the Maritime Domain in the Field of VHF Communication","date":"2023-06-01","arxiv_id":"2306.00614","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-an-unadaptable-asr-system","title":"Adapting an Unadaptable ASR System","date":"2023-06-01","arxiv_id":"2306.01208","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-contextual-biasing-for-transducer","title":"Adaptive Contextual Biasing for Transducer Based Streaming Speech Recognition","date":"2023-06-01","arxiv_id":"2306.00804","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrinames-most-asr-models-butcher-african","title":"AfriNames: Most ASR models \"butcher\" African Names","date":"2023-06-01","arxiv_id":"2306.00253","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-data-augmentation-for-domain","title":"Automatic Data Augmentation for Domain Adapted Fine-Tuning of Self-Supervised Speech Representations","date":"2023-06-01","arxiv_id":"2306.00481","repositories_listed":0,"syntology":null},{"url":null,"slug":"bypass-temporal-classification-weakly","title":"Bypass Temporal Classification: Weakly Supervised Automatic Speech Recognition with Imperfect Transcripts","date":"2023-06-01","arxiv_id":"2306.01031","repositories_listed":0,"syntology":null},{"url":null,"slug":"encoder-decoder-multimodal-speaker-change","title":"Encoder-decoder multimodal speaker change detection","date":"2023-06-01","arxiv_id":"2306.00680","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-the-unified-streaming-and-non","title":"Enhancing the Unified Streaming and Non-streaming Model with Contrastive Learning","date":"2023-06-01","arxiv_id":"2306.00755","repositories_listed":0,"syntology":null},{"url":null,"slug":"inspecting-spoken-language-understanding-from","title":"Inspecting Spoken Language Understanding from Kids for Basic Math Learning at Home","date":"2023-06-01","arxiv_id":"2306.00482","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-robustness-of-arabic-speech-dialect","title":"On the Robustness of Arabic Speech Dialect Identification","date":"2023-06-01","arxiv_id":"2306.03789","repositories_listed":0,"syntology":null},{"url":null,"slug":"some-voices-are-too-common-building-fair","title":"Some voices are too common: Building fair speech recognition systems using the Common Voice dataset","date":"2023-06-01","arxiv_id":"2306.03773","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-inpainting-context-based-speech","title":"Speech inpainting: Context-based speech synthesis guided by video","date":"2023-06-01","arxiv_id":"2306.00489","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-hate-speech-detection-in-low-resource","title":"Towards hate speech detection in low-resource languages: Comparing ASR to acoustic word embeddings on Wolof and Swahili","date":"2023-06-01","arxiv_id":"2306.00410","repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-and-structured-pruning-for-efficient","title":"Accurate and Structured Pruning for Efficient Automatic Speech Recognition","date":"2023-05-31","arxiv_id":"2305.19549","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-yet-effective-code-switching-language","title":"Simple yet Effective Code-Switching Language Identification with Multitask Pre-Training and Transfer Learning","date":"2023-05-31","arxiv_id":"2305.19759","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategies-for-improving-low-resource-speech","title":"Strategies for improving low resource speech to text translation relying on pre-trained ASR models","date":"2023-05-31","arxiv_id":"2306.00208","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-tag-team-approach-leveraging-cls-and","title":"The Tag-Team Approach: Leveraging CLS and Language Tagging for Enhancing Multilingual ASR","date":"2023-05-31","arxiv_id":"2305.19584","repositories_listed":0,"syntology":null},{"url":null,"slug":"vilas-integrating-vision-and-language-into","title":"VILAS: Exploring the Effects of Vision and Language Context in Automatic Speech Recognition","date":"2023-05-31","arxiv_id":"2305.19972","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-automatic-pronunciation-assessment","title":"Zero-Shot Automatic Pronunciation Assessment","date":"2023-05-31","arxiv_id":"2305.19563","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-multi-lingual-asr-models-for","title":"Adapting Multi-Lingual ASR Models for Handling Multiple Talkers","date":"2023-05-30","arxiv_id":"2305.18747","repositories_listed":0,"syntology":null},{"url":null,"slug":"stt4sg-350-a-speech-corpus-for-all-swiss","title":"STT4SG-350: A Speech Corpus for All Swiss German Dialect Regions","date":"2023-05-30","arxiv_id":"2305.18855","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-news-delivery-channel-recommendation","title":"The News Delivery Channel Recommendation Based on Granular Neural Network","date":"2023-05-30","arxiv_id":"2306.10022","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-selection-of-text-to-speech-data-to","title":"Towards Selection of Text-to-speech Data to Augment ASR Training","date":"2023-05-30","arxiv_id":"2306.00998","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-review-of-speaker-diarization","title":"An Experimental Review of Speaker Diarization methods with application to Two-Speaker Conversational Telephone Speech recordings","date":"2023-05-29","arxiv_id":"2305.18074","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-accurate-low-latency-asr-for","title":"Building Accurate Low Latency ASR for Streaming Voice Search","date":"2023-05-29","arxiv_id":"2305.18596","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-trust-explainable-ai-methods-on-asr-an","title":"Can We Trust Explainable AI Methods on ASR? An Evaluation on Phoneme Recognition","date":"2023-05-29","arxiv_id":"2305.18011","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-textless-spoken-language","title":"Improving Textless Spoken Language Understanding with Discrete Units as Intermediate Target","date":"2023-05-29","arxiv_id":"2305.18096","repositories_listed":0,"syntology":null},{"url":null,"slug":"retraining-free-customized-asr-for-enharmonic","title":"Retraining-free Customized ASR for Enharmonic Words Based on a Named-Entity-Aware Model and Phoneme Similarity Estimation","date":"2023-05-29","arxiv_id":"2305.17846","repositories_listed":0,"syntology":null},{"url":null,"slug":"rasr2-the-rwth-asr-toolkit-for-generic","title":"RASR2: The RWTH ASR Toolkit for Generic Sequence-to-sequence Speech Recognition","date":"2023-05-28","arxiv_id":"2305.17782","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-overview-and-comparative","title":"A Comprehensive Overview and Comparative Analysis on Deep Learning Models: CNN, RNN, LSTM, GRU","date":"2023-05-27","arxiv_id":"2305.17473","repositories_listed":0,"syntology":null},{"url":null,"slug":"2-bit-conformer-quantization-for-automatic","title":"2-bit Conformer quantization for automatic speech recognition","date":"2023-05-26","arxiv_id":"2305.16619","repositories_listed":0,"syntology":null},{"url":null,"slug":"disfluencyfixer-a-tool-to-enhance-language","title":"DisfluencyFixer: A tool to enhance Language Learning through Speech To Speech Disfluency Correction","date":"2023-05-26","arxiv_id":"2305.16957","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-multi-source-mt-to","title":"Robustness of Multi-Source MT to Transcription Errors","date":"2023-05-26","arxiv_id":"2305.16894","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-and-emotional-speech-a-word-level","title":"ASR and Emotional Speech: A Word-Level Investigation of the Mutual Impact of Speech and Emotion Recognition","date":"2023-05-25","arxiv_id":"2305.16065","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-scheduled-sampling-for-neural","title":"Improving Scheduled Sampling for Neural Transducer-based ASR","date":"2023-05-25","arxiv_id":"2305.15958","repositories_listed":0,"syntology":null},{"url":null,"slug":"intapt-information-theoretic-adversarial","title":"INTapt: Information-Theoretic Adversarial Prompt Tuning for Enhanced Non-Native Speech Recognition","date":"2023-05-25","arxiv_id":"2305.16371","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-neural-transducer","title":"Knowledge Distillation for Neural Transducer-based Target-Speaker ASR: Exploiting Parallel Mixture/Single-Talker Speech Data","date":"2023-05-25","arxiv_id":"2305.15971","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-expert-conformer-for-streaming","title":"Mixture-of-Expert Conformer for Streaming Multilingual ASR","date":"2023-05-25","arxiv_id":"2305.15663","repositories_listed":0,"syntology":null},{"url":null,"slug":"persistent-laplacian-enhanced-algorithm-for","title":"Persistent Laplacian-enhanced Algorithm for Scarcely Labeled Data Classification","date":"2023-05-25","arxiv_id":"2305.16239","repositories_listed":0,"syntology":null},{"url":null,"slug":"svarah-evaluating-english-asr-systems-on","title":"Svarah: Evaluating English ASR Systems on Indian Accents","date":"2023-05-25","arxiv_id":"2305.15760","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-modeling-of-multi-talker-overlapped","title":"Unified Modeling of Multi-Talker Overlapped Speech Recognition and Diarization with a Sidecar Separator","date":"2023-05-25","arxiv_id":"2305.16263","repositories_listed":0,"syntology":null},{"url":null,"slug":"viola-unified-codec-language-models-for","title":"VioLA: Unified Codec Language Models for Speech Recognition, Synthesis, and Translation","date":"2023-05-25","arxiv_id":"2305.16107","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-speech-pre-training-a-case","title":"Weakly-Supervised Speech Pre-training: A Case Study on Target Speech Recognition","date":"2023-05-25","arxiv_id":"2305.16286","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-ultrasound-tongue-images-for","title":"Incorporating Ultrasound Tongue Images for Audio-Visual Speech Enhancement through Knowledge Distillation","date":"2023-05-24","arxiv_id":"2305.14933","repositories_listed":0,"syntology":null},{"url":null,"slug":"interformer-interactive-local-and-global","title":"InterFormer: Interactive Local and Global Features Fusion for Automatic Speech Recognition","date":"2023-05-24","arxiv_id":"2305.16342","repositories_listed":0,"syntology":null},{"url":null,"slug":"iteratively-improving-speech-recognition-and","title":"Iteratively Improving Speech Recognition and Voice Conversion","date":"2023-05-24","arxiv_id":"2305.15055","repositories_listed":0,"syntology":null},{"url":null,"slug":"lms-with-a-voice-spoken-language-modeling","title":"Spoken Question Answering and Speech Continuation Using Spectrogram-Powered LLM","date":"2023-05-24","arxiv_id":"2305.15255","repositories_listed":0,"syntology":null},{"url":null,"slug":"rand-robustness-aware-norm-decay-for","title":"RAND: Robustness Aware Norm Decay For Quantized Seq2seq Models","date":"2023-05-24","arxiv_id":"2305.15536","repositories_listed":0,"syntology":null},{"url":null,"slug":"ba-sot-boundary-aware-serialized-output","title":"BA-SOT: Boundary-Aware Serialized Output Training for Multi-Talker ASR","date":"2023-05-23","arxiv_id":"2305.13716","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-knowledge-transfer-and","title":"Cross-lingual Knowledge Transfer and Iterative Pseudo-labeling for Low-Resource Speech Recognition with Transducers","date":"2023-05-23","arxiv_id":"2305.13652","repositories_listed":0,"syntology":null}],"record_sha256":"4df1bc07b9af2fcee004ab96c23882a1ef24941a1b688d3185f2008dcd7f2ac5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}