{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/15","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":31,"rows_per_page":100,"rows":[1401,1500],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/14","next":"/task/automatic-speech-recognition/papers/16","papers":[{"url":null,"slug":"filter-and-evolve-progressive-pseudo-label","title":"Filter and evolve: progressive pseudo label refining for semi-supervised automatic speech recognition","date":"2022-10-28","arxiv_id":"2210.16318","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-short-video-speech-recognition","title":"Random Utterance Concatenation Based Data Augmentation for Improving Short-video Speech Recognition","date":"2022-10-28","arxiv_id":"2210.15876","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-utterance-training-for-automatic","title":"Contextual-Utterance Training for Automatic Speech Recognition","date":"2022-10-27","arxiv_id":"2210.16238","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-effective-distillation-of-self","title":"Exploring Effective Distillation of Self-Supervised Speech Models for Automatic Speech Recognition","date":"2022-10-27","arxiv_id":"2210.15631","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-more-of-your-data-minimal-effort-data","title":"Make More of Your Data: Minimal Effort Data Augmentation for Automatic Speech Recognition and Translation","date":"2022-10-27","arxiv_id":"2210.15398","repositories_listed":0,"syntology":null},{"url":null,"slug":"san-a-robust-end-to-end-asr-model","title":"SAN: a robust end-to-end ASR model architecture","date":"2022-10-27","arxiv_id":"2210.15285","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulating-realistic-speech-overlaps-improves","title":"Simulating realistic speech overlaps improves multi-talker ASR","date":"2022-10-27","arxiv_id":"2210.15715","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-voice-conversion-via-intermediate","title":"Streaming Voice Conversion Via Intermediate Bottleneck Features And Non-streaming Teacher Guidance","date":"2022-10-27","arxiv_id":"2210.15158","repositories_listed":0,"syntology":null},{"url":null,"slug":"trscore-a-novel-gpt-based-readability-scorer","title":"TRScore: A Novel GPT-based Readability Scorer for ASR Segmentation and Punctuation model evaluation and selection","date":"2022-10-27","arxiv_id":"2210.15104","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-cloak-intelligibility-naturalness-timbre","title":"V-Cloak: Intelligibility-, Naturalness- & Timbre-Preserving Real-Time Voice Anonymization","date":"2022-10-27","arxiv_id":"2210.15140","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtuoso-massive-multilingual-speech-text","title":"Virtuoso: Massive Multilingual Speech-Text Joint Semi-Supervised Learning for Text-To-Speech","date":"2022-10-27","arxiv_id":"2210.15447","repositories_listed":0,"syntology":null},{"url":null,"slug":"weight-averaging-a-simple-yet-effective","title":"Weight Averaging: A Simple Yet Effective Method to Overcome Catastrophic Forgetting in Automatic Speech Recognition","date":"2022-10-27","arxiv_id":"2210.15282","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-use-of-large-pre-trained-models-for","title":"Efficient Utilization of Large Pre-Trained Models for Low Resource ASR","date":"2022-10-26","arxiv_id":"2210.15445","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-to-intent-prediction-to","title":"End-to-End Speech to Intent Prediction to improve E-commerce Customer Support Voicebot in Hindi and English","date":"2022-10-26","arxiv_id":"2211.07710","repositories_listed":0,"syntology":null},{"url":null,"slug":"four-in-one-a-joint-approach-to-inverse-text","title":"Four-in-One: A Joint Approach to Inverse Text Normalization, Punctuation, Capitalization, and Disfluency for Automatic Speech Recognition","date":"2022-10-26","arxiv_id":"2210.15063","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-speech-segmentation-using-acousto","title":"Smart Speech Segmentation using Acousto-Linguistic Features with look-ahead","date":"2022-10-26","arxiv_id":"2210.14446","repositories_listed":0,"syntology":null},{"url":null,"slug":"ufo2-a-unified-pre-training-framework-for","title":"UFO2: A unified pre-training framework for online and offline speech recognition","date":"2022-10-26","arxiv_id":"2210.14515","repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistic-enhanced-transformer-with-ctc","title":"Linguistic-Enhanced Transformer with CTC Embedding for Speech Recognition","date":"2022-10-25","arxiv_id":"2210.14725","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-effect-of-domain-selection","title":"Investigating self-supervised, weakly supervised and fully supervised training approaches for multi-domain automatic speech recognition: a study on Bangladeshi Bangla","date":"2022-10-24","arxiv_id":"2210.12921","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-domain-speech-enhancement-for-robust","title":"Time-Domain Speech Enhancement for Robust Automatic Speech Recognition","date":"2022-10-24","arxiv_id":"2210.13318","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-contrastive-self-supervised-pre","title":"Guided contrastive self-supervised pre-training for automatic speech recognition","date":"2022-10-22","arxiv_id":"2210.12335","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-visual-context-improve-automatic-speech","title":"Can Visual Context Improve Automatic Speech Recognition for an Embodied Agent?","date":"2022-10-21","arxiv_id":"2210.13189","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-bilingual-neural-transducer-with","title":"Optimizing Bilingual Neural Transducer with Synthetic Code-switching Text Generation","date":"2022-10-21","arxiv_id":"2210.12214","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-semi-supervised-end-to-end","title":"Improving Semi-supervised End-to-end Automatic Speech Recognition using CycleGAN and Inter-domain Losses","date":"2022-10-20","arxiv_id":"2210.11642","repositories_listed":0,"syntology":null},{"url":"/paper/g-augment-searching-for-the-meta-structure-of","slug":"g-augment-searching-for-the-meta-structure-of","title":"G-Augment: Searching for the Meta-Structure of Data Augmentation Policies for ASR","date":"2022-10-19","arxiv_id":"2210.10879","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-pseudo-labeling-from-the-start","title":"Continuous Pseudo-Labeling from the Start","date":"2022-10-17","arxiv_id":"2210.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-agnostic-code-switching-in-end-to","title":"Language-agnostic Code-Switching in Sequence-To-Sequence Speech Recognition","date":"2022-10-17","arxiv_id":"2210.08992","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-8-bit-quantization-for-on-device-speech","title":"Sub-8-bit quantization for on-device speech recognition: a regularization-free approach","date":"2022-10-17","arxiv_id":"2210.09188","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-jointly-transcribe-and-subtitle","title":"Learning to Jointly Transcribe and Subtitle for End-to-End Spontaneous Speech Recognition","date":"2022-10-14","arxiv_id":"2210.07771","repositories_listed":0,"syntology":null},{"url":null,"slug":"levoice-asr-systems-for-the-iscslp-2022","title":"LeVoice ASR Systems for the ISCSLP 2022 Intelligent Cockpit Speech Recognition Challenge","date":"2022-10-14","arxiv_id":"2210.07749","repositories_listed":0,"syntology":null},{"url":null,"slug":"hubert-tr-reviving-turkish-automatic-speech","title":"Experiments on Turkish ASR with Self-Supervised Speech Representation Learning","date":"2022-10-13","arxiv_id":"2210.07323","repositories_listed":0,"syntology":null},{"url":null,"slug":"summary-on-the-iscslp-2022-chinese-english","title":"Summary on the ISCSLP 2022 Chinese-English Code-Switching ASR Challenge","date":"2022-10-12","arxiv_id":"2210.06091","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-study-on-private-aggregation","title":"An Experimental Study on Private Aggregation of Teacher Ensemble Learning for End-to-End Speech Recognition","date":"2022-10-11","arxiv_id":"2210.05614","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-of-low-resource","title":"Automatic Speech Recognition of Low-Resource Languages Based on Chukchi","date":"2022-10-11","arxiv_id":"2210.05726","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-soft-and-hard-target-rnn-t","title":"Comparison of Soft and Hard Target RNN-T Distillation for Large-scale ASR","date":"2022-10-11","arxiv_id":"2210.05793","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-alignments-improve-autoregressive","title":"CTC Alignments Improve Autoregressive Translation","date":"2022-10-11","arxiv_id":"2210.05200","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-deliberation-for-multilingual-asr","title":"Scaling Up Deliberation for Multilingual ASR","date":"2022-10-11","arxiv_id":"2210.05785","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-punctuation-for-long-form-dictation","title":"Streaming Punctuation for Long-form Dictation with Transformers","date":"2022-10-11","arxiv_id":"2210.05756","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloud-based-automatic-speech-recognition","title":"Cloud-based Automatic Speech Recognition Systems for Southeast Asian Languages","date":"2022-10-07","arxiv_id":"2210.03580","repositories_listed":0,"syntology":null},{"url":null,"slug":"pronunciation-modeling-of-foreign-words-for","title":"Pronunciation Modeling of Foreign Words for Mandarin ASR by Considering the Effect of Language Transfer","date":"2022-10-07","arxiv_id":"2210.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"damage-control-during-domain-adaptation-for","title":"Damage Control During Domain Adaptation for Transducer Based Automatic Speech Recognition","date":"2022-10-06","arxiv_id":"2210.03255","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-acoustic-feature-transformation-in","title":"Efficient acoustic feature transformation in mismatched environments using a Guided-GAN","date":"2022-10-03","arxiv_id":"2210.00721","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-train-a-language-model-inside-an-end","title":"Can We Train a Language Model Inside an End-to-End ASR Model? - Investigating Effective Implicit Language Modeling","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-code-switched-asr-with-linguistic","title":"Improving Code-switched ASR with Linguistic Information","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-impact-of-asr-errors-on","title":"Investigating the Impact of ASR Errors on Spoken Implicit Discourse Relation Recognition","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"keyphrase-prediction-from-video-transcripts","title":"Keyphrase Prediction from Video Transcripts: New Dataset and Directions","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"language-specific-effects-on-automatic-speech","title":"Language-specific Effects on Automatic Speech Recognition Errors for World Englishes","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-progressive-compression-of","title":"Multi-stage Progressive Compression of Conformer Transducer for On-device Speech Recognition","date":"2022-10-01","arxiv_id":"2210.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-disfluency-detection-for-indian","title":"Zero-shot Disfluency Detection for Indian Languages","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-sparse-and-monotonic-attention-for","title":"Adaptive Sparse and Monotonic Attention for Transformer-based Automatic Speech Recognition","date":"2022-09-30","arxiv_id":"2209.15176","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-signal-dereverberation-for-machine","title":"Blind Signal Dereverberation for Machine Speech Recognition","date":"2022-09-30","arxiv_id":"2210.00117","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-performant-named-entity","title":"An Effective, Performant Named Entity Recognition System for Noisy Business Telephone Conversation Transcripts","date":"2022-09-27","arxiv_id":"2209.13736","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-for-speech-1","title":"Unsupervised domain adaptation for speech recognition with unsupervised error correction","date":"2022-09-24","arxiv_id":"2209.12043","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-asr-model-quality-on-disordered","title":"Assessing ASR Model Quality on Disordered Speech using BERTScore","date":"2022-09-21","arxiv_id":"2209.10591","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-parallel-voice-conversion-for-asr","title":"Non-Parallel Voice Conversion for ASR Augmentation","date":"2022-09-15","arxiv_id":"2209.06987","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-universally-deployable-asr-frontend-for","title":"A Universally-Deployable ASR Frontend for Joint Acoustic Echo Cancellation, Speech Enhancement, and Voice Separation","date":"2022-09-14","arxiv_id":"2209.06410","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-pruning-improving-neural-network","title":"Federated Pruning: Improving Neural Network Efficiency with Federated Learning","date":"2022-09-14","arxiv_id":"2209.06359","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-self-attention-head-diversity-for","title":"Analysis of Self-Attention Head Diversity for Conformer-based Automatic Speech Recognition","date":"2022-09-13","arxiv_id":"2209.06096","repositories_listed":0,"syntology":null},{"url":null,"slug":"bangla-wave-improving-bangla-automatic-speech","title":"Bangla-Wave: Improving Bangla Automatic Speech Recognition Utilizing N-gram Language Models","date":"2022-09-13","arxiv_id":"2209.12650","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-asr-pathways-a-sparse-multilingual","title":"Learning ASR pathways: A sparse multilingual ASR model","date":"2022-09-13","arxiv_id":"2209.05735","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-end-to-end-multilingual-speech","title":"Streaming End-to-End Multilingual Speech Recognition with Joint Language Identification","date":"2022-09-13","arxiv_id":"2209.06058","repositories_listed":0,"syntology":null},{"url":null,"slug":"vararray-meets-t-sot-advancing-the-state-of","title":"VarArray Meets t-SOT: Advancing the State of the Art of Streaming Distant Conversational Speech Recognition","date":"2022-09-12","arxiv_id":"2209.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexicon-and-attention-based-handwritten-text","title":"Lexicon and Attention based Handwritten Text Recognition System","date":"2022-09-11","arxiv_id":"2209.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-end-to-end-target-speaker-asr","title":"Streaming Target-Speaker ASR with Neural Transducer","date":"2022-09-09","arxiv_id":"2209.04175","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-dependent-structure-for-utterances","title":"Modeling Dependent Structure for Utterances in ASR Evaluation","date":"2022-09-07","arxiv_id":"2209.05281","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-the-knowledge-of-bert-for-ctc","title":"Distilling the Knowledge of BERT for CTC-based ASR","date":"2022-09-05","arxiv_id":"2209.02030","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-enhanced-citrinet-for-speech","title":"Attention Enhanced Citrinet for Speech Recognition","date":"2022-09-01","arxiv_id":"2209.00261","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepcon-an-end-to-end-multilingual-toolkit","title":"DeepCon: An End-to-End Multilingual Toolkit for Automatic Minuting of Multi-Party Dialogues","date":"2022-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-automatic-speech-recognition","title":"Evaluation of Automatic Speech Recognition for Conversational Speech in Dutch, English and German: What Goes Missing?","date":"2022-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-translation-of-french-live-speech","title":"Robust Translation of French Live Speech Transcripts","date":"2022-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-language-agnostic-multilingual-streaming-on","title":"A Language Agnostic Multilingual Streaming On-Device ASR System","date":"2022-08-29","arxiv_id":"2208.13916","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-data-partitioning-strategies","title":"Investigating data partitioning strategies for crosslinguistic low-resource ASR evaluation","date":"2022-08-26","arxiv_id":"2208.12888","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-disentangled-representations-all-you-need","title":"Are disentangled representations all you need to build speaker anonymization systems?","date":"2022-08-22","arxiv_id":"2208.10497","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualvoice-speech-interaction-that","title":"DualVoice: Speech Interaction that Discriminates between Normal and Whispered Voice Input","date":"2022-08-22","arxiv_id":"2208.10499","repositories_listed":0,"syntology":null},{"url":"/paper/building-a-public-domain-voice-database-for","slug":"building-a-public-domain-voice-database-for","title":"Building a Public Domain Voice Database for Odia","date":"2022-08-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-hypernasality-estimation-with","title":"Improving Hypernasality Estimation with Automatic Speech Recognition in Cleft Palate Speech","date":"2022-08-10","arxiv_id":"2208.05122","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-vocabulary-speech-recognition-for","title":"Large vocabulary speech recognition for languages of Africa: multilingual modeling and self-supervised learning","date":"2022-08-05","arxiv_id":"2208.03067","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-on-asr-systems-an","title":"Adversarial Attacks on ASR Systems: An Overview","date":"2022-08-03","arxiv_id":"2208.02250","repositories_listed":0,"syntology":null},{"url":"/paper/automatic-speech-recognition-in-german-a","slug":"automatic-speech-recognition-in-german-a","title":"Automatic Speech Recognition in German: A Detailed Error Analysis","date":"2022-08-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-disparities-between-accents-in","title":"Global Performance Disparities Between English-Language Accents in Automatic Speech Recognition","date":"2022-08-01","arxiv_id":"2208.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-hypothesis-rnn-t-loss-for","title":"Multiple-hypothesis RNN-T Loss for Unsupervised Fine-tuning and Self-training of Neural Transducer","date":"2022-07-29","arxiv_id":"2207.14736","repositories_listed":0,"syntology":null},{"url":null,"slug":"pronunciation-aware-unique-character-encoding","title":"Pronunciation-aware unique character encoding for RNN Transducer-based Mandarin speech recognition","date":"2022-07-29","arxiv_id":"2207.14578","repositories_listed":0,"syntology":null},{"url":null,"slug":"thutmose-tagger-single-pass-neural-model-for","title":"Thutmose Tagger: Single-pass neural model for Inverse Text Normalization","date":"2022-07-29","arxiv_id":"2208.00064","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-driven-subword-grammar-modeling-for","title":"Knowledge-driven Subword Grammar Modeling for Automatic Speech Recognition in Tamil and Kannada","date":"2022-07-27","arxiv_id":"2207.13333","repositories_listed":0,"syntology":null},{"url":null,"slug":"subword-dictionary-learning-and-segmentation","title":"Subword Dictionary Learning and Segmentation Techniques for Automatic Speech Recognition in Tamil and Kannada","date":"2022-07-27","arxiv_id":"2207.13331","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-dual-mode-speech-recognition-model","title":"Learning a Dual-Mode Speech Recognition Model via Self-Pruning","date":"2022-07-25","arxiv_id":"2207.11906","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-data-selection-for-speech","title":"Unsupervised data selection for Speech Recognition with contrastive loss ratios","date":"2022-07-25","arxiv_id":"2207.12028","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-error-detection-via-audio-transcript","title":"ASR Error Detection via Audio-Transcript entailment","date":"2022-07-22","arxiv_id":"2207.10849","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-data-driven-inverse-text","title":"Improving Data Driven Inverse Text Normalization using Data Augmentation","date":"2022-07-20","arxiv_id":"2207.09674","repositories_listed":0,"syntology":null},{"url":null,"slug":"ilasr-privacy-preserving-incremental-learning","title":"ILASR: Privacy-Preserving Incremental Learning for Automatic Speech Recognition at Production Scale","date":"2022-07-19","arxiv_id":"2207.09078","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-spoken-language-understanding-3","title":"End-to-End Spoken Language Understanding: Performance analyses of a voice command task in a low resource setting","date":"2022-07-17","arxiv_id":"2207.08179","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-geographic-disparities-in-automatic","title":"Reducing Geographic Disparities in Automatic Speech Recognition via Elastic Weight Consolidation","date":"2022-07-16","arxiv_id":"2207.07850","repositories_listed":0,"syntology":null},{"url":"/paper/direction-aware-joint-adaptation-of-neural","slug":"direction-aware-joint-adaptation-of-neural","title":"Direction-Aware Joint Adaptation of Neural Speech Enhancement and Recognition in Real Multiparty Conversational Environments","date":"2022-07-15","arxiv_id":"2207.07273","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-low-resource-quechua","title":"Data Augmentation for Low-Resource Quechua ASR Improvement","date":"2022-07-14","arxiv_id":"2207.06872","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-versus-wide-an-analysis-of-student","title":"Deep versus Wide: An Analysis of Student Architectures for Task-Agnostic Knowledge Distillation of Self-Supervised Speech Models","date":"2022-07-14","arxiv_id":"2207.06867","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-recognition-modeling-from","title":"End-to-end speech recognition modeling from de-identified data","date":"2022-07-12","arxiv_id":"2207.05469","repositories_listed":0,"syntology":null},{"url":null,"slug":"huqariq-a-multilingual-speech-corpus-of","title":"Huqariq: A Multilingual Speech Corpus of Native Languages of Peru for Speech Recognition","date":"2022-07-12","arxiv_id":"2207.05498","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-continual-learning-of-end-to-end","title":"Online Continual Learning of End-to-End Speech Recognition Models","date":"2022-07-11","arxiv_id":"2207.05071","repositories_listed":0,"syntology":null},{"url":null,"slug":"pmct-patched-multi-condition-training-for","title":"pMCT: Patched Multi-Condition Training for Robust Speech Recognition","date":"2022-07-11","arxiv_id":"2207.04949","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-to-punctuated-text","title":"End-to-end Speech-to-Punctuated-Text Recognition","date":"2022-07-07","arxiv_id":"2207.03169","repositories_listed":0,"syntology":null}],"record_sha256":"0479e5acda1caba07e41634e7b0f65846e36ac6f78d23b7ba7475ef32ffe7158","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}