{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/15","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":32,"rows_per_page":100,"rows":[1401,1500],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/14","next":"/task/automatic-speech-recognition-2/papers/16","papers":[{"url":null,"slug":"accelerating-transducers-through-adjacent","title":"Accelerating Transducers through Adjacent Token Merging","date":"2023-06-28","arxiv_id":"2306.16009","repositories_listed":0,"syntology":null},{"url":null,"slug":"master-asr-achieving-multilingual-scalability","title":"Master-ASR: Achieving Multilingual Scalability and Low-Resource Adaptation in ASR with Modular Learning","date":"2023-06-23","arxiv_id":"2306.15686","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-chime-7-dasr-challenge-distant-meeting","title":"The CHiME-7 DASR Challenge: Distant Meeting Transcription with Multiple Devices in Diverse Scenarios","date":"2023-06-23","arxiv_id":"2306.13734","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-role-of-audio-in-video","title":"Exploring the Role of Audio in Video Captioning","date":"2023-06-21","arxiv_id":"2306.12559","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-self-learning-with-weak-supervision","title":"Federated Self-Learning with Weak Supervision for Speech Recognition","date":"2023-06-21","arxiv_id":"2306.12015","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-when-to-trust-which-teacher-for","title":"Learning When to Trust Which Teacher for Weakly Supervised ASR","date":"2023-06-21","arxiv_id":"2306.12012","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-encoder-for-joint-speech-separation","title":"Mixture Encoder for Joint Speech Separation and Recognition","date":"2023-06-21","arxiv_id":"2306.12173","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategies-in-transfer-learning-for-low","title":"Strategies in Transfer Learning for Low-Resource Speech Synthesis: Phone Mapping, Features Input, and Source Language Selection","date":"2023-06-21","arxiv_id":"2306.12040","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexical-speaker-error-correction-leveraging","title":"Lexical Speaker Error Correction: Leveraging Language Models for Speaker Diarization Error Correction","date":"2023-06-15","arxiv_id":"2306.09313","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobileasr-a-resource-aware-on-device","title":"MobileASR: A resource-aware on-device learning framework for user voice personalization applications on mobile phones","date":"2023-06-15","arxiv_id":"2306.09384","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-code-switching-and-named-entity","title":"Improving Code-Switching and Named Entity Recognition in ASR with Speech Editing based Data Augmentation","date":"2023-06-14","arxiv_id":"2306.08588","repositories_listed":0,"syntology":null},{"url":null,"slug":"dctx-conformer-dynamic-context-carry-over-for","title":"DCTX-Conformer: Dynamic context carry-over for low latency unified streaming and non-streaming Conformer ASR","date":"2023-06-13","arxiv_id":"2306.08175","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-beamformer-exploiting-non","title":"Statistical Beamformer Exploiting Non-stationarity and Sparsity with Spatially Constrained ICA for Robust Speech Recognition","date":"2023-06-13","arxiv_id":"2306.07562","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-frequency-attention-alternative-to","title":"Multi-View Frequency-Attention Alternative to CNN Frontends for Automatic Speech Recognition","date":"2023-06-12","arxiv_id":"2306.06954","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-audio-textual-architecture-for-1","title":"Multimodal Audio-textual Architecture for Robust Spoken Language Understanding","date":"2023-06-12","arxiv_id":"2306.06819","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-n-gram-approximation-of-pre-trained","title":"On the N-gram Approximation of Pre-trained Language Models","date":"2023-06-12","arxiv_id":"2306.06892","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-experiencing-misrecognition-by","title":"Impact of Experiencing Misrecognition by Teachable Agents on Learning and Rapport","date":"2023-06-11","arxiv_id":"2306.07302","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-can-an-accent-identifier-learn-probing","title":"What Can an Accent Identifier Learn? Probing Phonetic and Prosodic Information in a Wav2vec2-based Accent Identification Model","date":"2023-06-10","arxiv_id":"2306.06524","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-frame-level-classifier-for-word","title":"Improving Frame-level Classifier for Word Timings with Non-peaky CTC in End-to-End Automatic Speech Recognition","date":"2023-06-09","arxiv_id":"2306.07949","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-language-model-integration-for","title":"Improving Language Model Integration for Neural Machine Translation","date":"2023-06-08","arxiv_id":"2306.05077","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-reliability-of-automatic","title":"A study on the impact of Self-Supervised Learning on automatic dysarthric speech assessment","date":"2023-06-07","arxiv_id":"2306.04337","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-asr-based-tutor-for-learning-to-read-how","title":"An ASR-Based Tutor for Learning to Read: How to Optimize Feedback to First Graders","date":"2023-06-07","arxiv_id":"2306.04190","repositories_listed":0,"syntology":null},{"url":null,"slug":"fooctts-generating-arabic-speech-with","title":"FOOCTTS: Generating Arabic Speech with Acoustic Environment for Football Commentator","date":"2023-06-07","arxiv_id":"2306.07936","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-from-pre-trained-language","title":"Transfer Learning from Pre-trained Language Models Improves End-to-End Speech Summarization","date":"2023-06-07","arxiv_id":"2306.04233","repositories_listed":0,"syntology":null},{"url":null,"slug":"alzheimer-disease-classification-through-asr","title":"Alzheimer Disease Classification through ASR-based Transcriptions: Exploring the Impact of Punctuation and Pauses","date":"2023-06-06","arxiv_id":"2306.03443","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-assessment-of-oral-reading-accuracy","title":"Automatic Assessment of Oral Reading Accuracy for Reading Diagnostics","date":"2023-06-06","arxiv_id":"2306.03444","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-fairness-and-robustness-in-end-to","title":"Improving Fairness and Robustness in End-to-End Speech Recognition through unsupervised clustering","date":"2023-06-06","arxiv_id":"2306.06083","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-l2-phonemes-using-articulatory","title":"Incorporating L2 Phonemes Using Articulatory Features for Robust Speech Recognition","date":"2023-06-05","arxiv_id":"2306.02534","repositories_listed":0,"syntology":null},{"url":null,"slug":"otf-optimal-transport-based-fusion-of","title":"OTF: Optimal Transport based Fusion of Supervised and Self-Supervised Learning Models for Automatic Speech Recognition","date":"2023-06-05","arxiv_id":"2306.02541","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-joint-target-and-non-target","title":"End-to-End Joint Target and Non-Target Speakers ASR","date":"2023-06-04","arxiv_id":"2306.02273","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-with-score","title":"Audio-Visual Speech Enhancement with Score-Based Generative Models","date":"2023-06-02","arxiv_id":"2306.01432","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-training-for-end-to-end-streaming","title":"Improved Training for End-to-End Streaming Automatic Speech Recognition Model with Punctuation","date":"2023-06-02","arxiv_id":"2306.01296","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-speech-to-confusion-network-speech","title":"Streaming Speech-to-Confusion Network Speech Recognition","date":"2023-06-02","arxiv_id":"2306.03778","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptation-and-optimization-of-automatic","title":"Adaptation and Optimization of Automatic Speech Recognition (ASR) for the Maritime Domain in the Field of VHF Communication","date":"2023-06-01","arxiv_id":"2306.00614","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrinames-most-asr-models-butcher-african","title":"AfriNames: Most ASR models \"butcher\" African Names","date":"2023-06-01","arxiv_id":"2306.00253","repositories_listed":0,"syntology":null},{"url":null,"slug":"bypass-temporal-classification-weakly","title":"Bypass Temporal Classification: Weakly Supervised Automatic Speech Recognition with Imperfect Transcripts","date":"2023-06-01","arxiv_id":"2306.01031","repositories_listed":0,"syntology":null},{"url":null,"slug":"encoder-decoder-multimodal-speaker-change","title":"Encoder-decoder multimodal speaker change detection","date":"2023-06-01","arxiv_id":"2306.00680","repositories_listed":0,"syntology":null},{"url":null,"slug":"inspecting-spoken-language-understanding-from","title":"Inspecting Spoken Language Understanding from Kids for Basic Math Learning at Home","date":"2023-06-01","arxiv_id":"2306.00482","repositories_listed":0,"syntology":null},{"url":null,"slug":"some-voices-are-too-common-building-fair","title":"Some voices are too common: Building fair speech recognition systems using the Common Voice dataset","date":"2023-06-01","arxiv_id":"2306.03773","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-hate-speech-detection-in-low-resource","title":"Towards hate speech detection in low-resource languages: Comparing ASR to acoustic word embeddings on Wolof and Swahili","date":"2023-06-01","arxiv_id":"2306.00410","repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-and-structured-pruning-for-efficient","title":"Accurate and Structured Pruning for Efficient Automatic Speech Recognition","date":"2023-05-31","arxiv_id":"2305.19549","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-yet-effective-code-switching-language","title":"Simple yet Effective Code-Switching Language Identification with Multitask Pre-Training and Transfer Learning","date":"2023-05-31","arxiv_id":"2305.19759","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategies-for-improving-low-resource-speech","title":"Strategies for improving low resource speech to text translation relying on pre-trained ASR models","date":"2023-05-31","arxiv_id":"2306.00208","repositories_listed":0,"syntology":null},{"url":null,"slug":"vilas-integrating-vision-and-language-into","title":"VILAS: Exploring the Effects of Vision and Language Context in Automatic Speech Recognition","date":"2023-05-31","arxiv_id":"2305.19972","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-automatic-pronunciation-assessment","title":"Zero-Shot Automatic Pronunciation Assessment","date":"2023-05-31","arxiv_id":"2305.19563","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-multi-lingual-asr-models-for","title":"Adapting Multi-Lingual ASR Models for Handling Multiple Talkers","date":"2023-05-30","arxiv_id":"2305.18747","repositories_listed":0,"syntology":null},{"url":null,"slug":"stt4sg-350-a-speech-corpus-for-all-swiss","title":"STT4SG-350: A Speech Corpus for All Swiss German Dialect Regions","date":"2023-05-30","arxiv_id":"2305.18855","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-selection-of-text-to-speech-data-to","title":"Towards Selection of Text-to-speech Data to Augment ASR Training","date":"2023-05-30","arxiv_id":"2306.00998","repositories_listed":0,"syntology":null},{"url":"/paper/a-hierarchical-context-aware-modeling","slug":"a-hierarchical-context-aware-modeling","title":"A Hierarchical Context-aware Modeling Approach for Multi-aspect and Multi-granular Pronunciation Assessment","date":"2023-05-29","arxiv_id":"2305.18146","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-accurate-low-latency-asr-for","title":"Building Accurate Low Latency ASR for Streaming Voice Search","date":"2023-05-29","arxiv_id":"2305.18596","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-trust-explainable-ai-methods-on-asr-an","title":"Can We Trust Explainable AI Methods on ASR? An Evaluation on Phoneme Recognition","date":"2023-05-29","arxiv_id":"2305.18011","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-textless-spoken-language","title":"Improving Textless Spoken Language Understanding with Discrete Units as Intermediate Target","date":"2023-05-29","arxiv_id":"2305.18096","repositories_listed":0,"syntology":null},{"url":null,"slug":"retraining-free-customized-asr-for-enharmonic","title":"Retraining-free Customized ASR for Enharmonic Words Based on a Named-Entity-Aware Model and Phoneme Similarity Estimation","date":"2023-05-29","arxiv_id":"2305.17846","repositories_listed":0,"syntology":null},{"url":null,"slug":"2-bit-conformer-quantization-for-automatic","title":"2-bit Conformer quantization for automatic speech recognition","date":"2023-05-26","arxiv_id":"2305.16619","repositories_listed":0,"syntology":null},{"url":null,"slug":"disfluencyfixer-a-tool-to-enhance-language","title":"DisfluencyFixer: A tool to enhance Language Learning through Speech To Speech Disfluency Correction","date":"2023-05-26","arxiv_id":"2305.16957","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-and-emotional-speech-a-word-level","title":"ASR and Emotional Speech: A Word-Level Investigation of the Mutual Impact of Speech and Emotion Recognition","date":"2023-05-25","arxiv_id":"2305.16065","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-scheduled-sampling-for-neural","title":"Improving Scheduled Sampling for Neural Transducer-based ASR","date":"2023-05-25","arxiv_id":"2305.15958","repositories_listed":0,"syntology":null},{"url":null,"slug":"intapt-information-theoretic-adversarial","title":"INTapt: Information-Theoretic Adversarial Prompt Tuning for Enhanced Non-Native Speech Recognition","date":"2023-05-25","arxiv_id":"2305.16371","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-expert-conformer-for-streaming","title":"Mixture-of-Expert Conformer for Streaming Multilingual ASR","date":"2023-05-25","arxiv_id":"2305.15663","repositories_listed":0,"syntology":null},{"url":null,"slug":"svarah-evaluating-english-asr-systems-on","title":"Svarah: Evaluating English ASR Systems on Indian Accents","date":"2023-05-25","arxiv_id":"2305.15760","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-modeling-of-multi-talker-overlapped","title":"Unified Modeling of Multi-Talker Overlapped Speech Recognition and Diarization with a Sidecar Separator","date":"2023-05-25","arxiv_id":"2305.16263","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-ultrasound-tongue-images-for","title":"Incorporating Ultrasound Tongue Images for Audio-Visual Speech Enhancement through Knowledge Distillation","date":"2023-05-24","arxiv_id":"2305.14933","repositories_listed":0,"syntology":null},{"url":null,"slug":"interformer-interactive-local-and-global","title":"InterFormer: Interactive Local and Global Features Fusion for Automatic Speech Recognition","date":"2023-05-24","arxiv_id":"2305.16342","repositories_listed":0,"syntology":null},{"url":null,"slug":"iteratively-improving-speech-recognition-and","title":"Iteratively Improving Speech Recognition and Voice Conversion","date":"2023-05-24","arxiv_id":"2305.15055","repositories_listed":0,"syntology":null},{"url":null,"slug":"ba-sot-boundary-aware-serialized-output","title":"BA-SOT: Boundary-Aware Serialized Output Training for Multi-Talker ASR","date":"2023-05-23","arxiv_id":"2305.13716","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-knowledge-transfer-and","title":"Cross-lingual Knowledge Transfer and Iterative Pseudo-labeling for Low-Resource Speech Recognition with Transducers","date":"2023-05-23","arxiv_id":"2305.13652","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-openai-s-whisper-asr-for","title":"Evaluating OpenAI's Whisper ASR for Punctuation Prediction and Topic Modeling of life histories of the Museum of the Person","date":"2023-05-23","arxiv_id":"2305.14580","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-meets-llm-a-novel-approach-to","title":"Graph Meets LLM: A Novel Approach to Collaborative Filtering for Robust Conversational Understanding","date":"2023-05-23","arxiv_id":"2305.14449","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-transferability-of-whisper-based","title":"On the Transferability of Whisper-based Representations for \"In-the-Wild\" Cross-Task Downstream Speech Applications","date":"2023-05-23","arxiv_id":"2305.14546","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-predictive-asr-for-latency","title":"Personalized Predictive ASR for Latency Reduction in Voice Assistants","date":"2023-05-23","arxiv_id":"2305.13794","repositories_listed":0,"syntology":null},{"url":null,"slug":"se-bridge-speech-enhancement-with-consistent","title":"SE-Bridge: Speech Enhancement with Consistent Brownian Bridge","date":"2023-05-23","arxiv_id":"2305.13796","repositories_listed":0,"syntology":null},{"url":null,"slug":"tranusr-phoneme-to-word-transcoder-based","title":"TranUSR: Phoneme-to-word Transcoder Based Unified Speech Representation Learning for Cross-lingual Speech Recognition","date":"2023-05-23","arxiv_id":"2305.13629","repositories_listed":0,"syntology":null},{"url":null,"slug":"debiased-automatic-speech-recognition-for","title":"Debiased Automatic Speech Recognition for Dysarthric Speech via Sample Reweighting with Sample Affinity Test","date":"2023-05-22","arxiv_id":"2305.13108","repositories_listed":0,"syntology":null},{"url":null,"slug":"gncformer-enhanced-self-attention-for","title":"GNCformer Enhanced Self-attention for Automatic Speech Recognition","date":"2023-05-22","arxiv_id":"2305.12755","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-generation-with-speech-synthesis-for-asr","title":"Text Generation with Speech Synthesis for ASR Data Augmentation","date":"2023-05-22","arxiv_id":"2305.16333","repositories_listed":0,"syntology":null},{"url":null,"slug":"casa-asr-context-aware-speaker-attributed-asr","title":"CASA-ASR: Context-Aware Speaker-Attributed ASR","date":"2023-05-21","arxiv_id":"2305.12459","repositories_listed":0,"syntology":null},{"url":null,"slug":"hystoc-obtaining-word-confidences-for-fusion","title":"Hystoc: Obtaining word confidences for fusion of end-to-end ASR systems","date":"2023-05-21","arxiv_id":"2305.12579","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-efficacy-and-noise-robustness-of","title":"On the Efficacy and Noise-Robustness of Jointly Learned Speech Emotion and Automatic Speech Recognition","date":"2023-05-21","arxiv_id":"2305.12540","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-vad-low-latency-voice-activity","title":"Semantic VAD: Low-Latency Voice Activity Detection for Speech Interaction","date":"2023-05-21","arxiv_id":"2305.12450","repositories_listed":0,"syntology":null},{"url":null,"slug":"vakta-setu-a-speech-to-speech-machine","title":"VAKTA-SETU: A Speech-to-Speech Machine Translation Service in Select Indic Languages","date":"2023-05-21","arxiv_id":"2305.12518","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-representations-in-speech","title":"Self-supervised representations in speech-based depression detection","date":"2023-05-20","arxiv_id":"2305.12263","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-asr-via-cross-lingual-pseudo","title":"Unsupervised ASR via Cross-Lingual Pseudo-Labeling","date":"2023-05-19","arxiv_id":"2305.13330","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lexical-aware-non-autoregressive","title":"A Lexical-aware Non-autoregressive Transformer-based ASR Model","date":"2023-05-18","arxiv_id":"2305.10839","repositories_listed":0,"syntology":null},{"url":null,"slug":"ml-superb-multilingual-speech-universal","title":"ML-SUPERB: Multilingual Speech Universal PERformance Benchmark","date":"2023-05-18","arxiv_id":"2305.10615","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-speaker-disentanglement-using","title":"Adversarial Speaker Disentanglement Using Unannotated External Data for Self-supervised Representation Based Voice Conversion","date":"2023-05-16","arxiv_id":"2305.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-agnostic-language-modeling-for-on","title":"Application-Agnostic Language Modeling for On-Device ASR","date":"2023-05-16","arxiv_id":"2305.09764","repositories_listed":0,"syntology":null},{"url":null,"slug":"critical-appraisal-of-artificial-intelligence","title":"Critical Appraisal of Artificial Intelligence-Mediated Communication","date":"2023-05-15","arxiv_id":"2305.11897","repositories_listed":0,"syntology":null},{"url":null,"slug":"ood-speech-a-large-bengali-speech-recognition","title":"OOD-Speech: A Large Bengali Speech Recognition Dataset for Out-of-Distribution Benchmarking","date":"2023-05-15","arxiv_id":"2305.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-neural-factor-analysis-for","title":"Self-supervised Neural Factor Analysis for Disentangling Utterance-level Speech Representations","date":"2023-05-14","arxiv_id":"2305.08099","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-for-end-to-end-asr-by","title":"Continual Learning for End-to-End ASR by Averaging Domain Experts","date":"2023-05-12","arxiv_id":"2305.09681","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-sensitivity-of-automatic","title":"Investigating the Sensitivity of Automatic Speech Recognition Systems to Phonetic Variation in L2 Englishes","date":"2023-05-12","arxiv_id":"2305.07389","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-audio-text-encoders-are-effective","title":"Masked Audio Text Encoders are Effective Multi-Modal Rescorers","date":"2023-05-11","arxiv_id":"2305.07677","repositories_listed":0,"syntology":null},{"url":null,"slug":"quran-recitation-recognition-using-end-to-end","title":"Quran Recitation Recognition using End-to-End Deep Learning","date":"2023-05-10","arxiv_id":"2305.07034","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-language-dependency-for","title":"Exploration of Language Dependency for Japanese Self-Supervised Speech Representation Models","date":"2023-05-09","arxiv_id":"2305.05201","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-acoustic-and-semantic-contextual","title":"Robust Acoustic and Semantic Contextual Biasing in Neural Transducers for Speech Recognition","date":"2023-05-09","arxiv_id":"2305.05271","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-needs-decoders-efficient-estimation-of","title":"Who Needs Decoders? Efficient Estimation of Sequence-level Attributes","date":"2023-05-09","arxiv_id":"2305.05098","repositories_listed":0,"syntology":null},{"url":"/paper/fast-conformer-with-linearly-scalable","slug":"fast-conformer-with-linearly-scalable","title":"Fast Conformer with Linearly Scalable Attention for Efficient Speech Recognition","date":"2023-05-08","arxiv_id":"2305.05084","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-temporal-lip-audio-memory-for-visual","title":"Multi-Temporal Lip-Audio Memory for Visual Speech Recognition","date":"2023-05-08","arxiv_id":"2305.04542","repositories_listed":0,"syntology":null},{"url":null,"slug":"lookahead-when-it-matters-adaptive-non-causal","title":"Lookahead When It Matters: Adaptive Non-causal Transformers for Streaming Neural Transducers","date":"2023-05-07","arxiv_id":"2305.04159","repositories_listed":0,"syntology":null},{"url":null,"slug":"employing-hybrid-deep-neural-networks-on-dari","title":"Employing Hybrid Deep Neural Networks on Dari Speech","date":"2023-05-04","arxiv_id":"2305.03200","repositories_listed":0,"syntology":null}],"record_sha256":"b27d94cf8c8ede9b11128d79e2d18f8f4220263a1ee7eb4353db1f5f2f86fae4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}