{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/36","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":36,"pages_in_order":65,"rows_per_page":100,"rows":[3501,3600],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/35","next":"/task/speech-recognition/papers/37","papers":[{"url":null,"slug":"learning-of-frequency-time-attention","title":"Learning of Time-Frequency Attention Mechanism for Automatic Modulation Recognition","date":"2021-11-05","arxiv_id":"2111.03258","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-teacher-towards-better-knowledge","title":"Oracle Teacher: Leveraging Target Information for Better Knowledge Distillation of CTC Models","date":"2021-11-05","arxiv_id":"2111.03664","repositories_listed":0,"syntology":null},{"url":"/paper/a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","slug":"a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","title":"A Fine-tuned Wav2vec 2.0/HuBERT Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding","date":"2021-11-04","arxiv_id":"2111.02735","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-for-air-traffic-control","title":"Speech recognition for air traffic control via feature learning and end-to-end training","date":"2021-11-04","arxiv_id":"2111.02654","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-can-improve-asr-in-very-low","title":"Voice Conversion Can Improve ASR in Very Low-Resource Settings","date":"2021-11-04","arxiv_id":"2111.02674","repositories_listed":0,"syntology":null},{"url":null,"slug":"stc-speaker-recognition-systems-for-the-nist","title":"STC speaker recognition systems for the NIST SRE 2021","date":"2021-11-03","arxiv_id":"2111.02298","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advances-in-end-to-end-automatic","title":"Recent Advances in End-to-End Automatic Speech Recognition","date":"2021-11-02","arxiv_id":"2111.01690","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-data-relabeling-for-robust-and","title":"Collaborative Data Relabeling for Robust and Diverse Voice Apps Recommendation in Intelligent Personal Assistants","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-grammatical-theories-of-code-mixing","title":"Comparing Grammatical Theories of Code-Mixing","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-punctuation-restoration-for","title":"Comprehensive Punctuation Restoration for English and Polish","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"indic-languages-automatic-speech-recognition","title":"Indic Languages Automatic Speech Recognition using Meta-Learning Approach","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-transduction-with-graph-based","title":"Sequence Transduction with Graph-based Supervision","date":"2021-11-01","arxiv_id":"2111.01272","repositories_listed":0,"syntology":null},{"url":null,"slug":"snri-target-training-for-joint-speech","title":"SNRi Target Training for Joint Speech Enhancement and Recognition","date":"2021-11-01","arxiv_id":"2111.00764","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-technology-for-everyone-automatic-1","title":"Speech Technology for Everyone: Automatic Speech Recognition for Non-Native English","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-query-auto-completion","title":"Voice Query Auto Completion","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-emotion-recognition-using-quaternion","title":"Speech Emotion Recognition Using Quaternion Convolutional Neural Networks","date":"2021-10-31","arxiv_id":"2111.00404","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-attention-conformer-for-context","title":"Cross-attention conformer for context modeling in speech enhancement for ASR","date":"2021-10-30","arxiv_id":"2111.00127","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-labeling-for-massively-multilingual","title":"Pseudo-Labeling for Massively Multilingual Speech Recognition","date":"2021-10-30","arxiv_id":"2111.00161","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-conditioning-of-acoustic-models-using","title":"Speaker conditioning of acoustic models using affine transformation for multi-speaker speech recognition","date":"2021-10-30","arxiv_id":"2111.00320","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-unsupervised-and-text-augmented","title":"Combining Unsupervised and Text Augmented Semi-Supervised Learning for Low Resourced Autoregressive Speech Recognition","date":"2021-10-29","arxiv_id":"2110.15836","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-asr-outputs-in-joint-training-for","title":"Fusing ASR Outputs in Joint Training for Speech Emotion Recognition","date":"2021-10-29","arxiv_id":"2110.15684","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-speech-separation-with-recurrent","title":"Continuous Speech Separation with Recurrent Selective Attention Network","date":"2021-10-28","arxiv_id":"2110.14838","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-noise-robustness-of-contrastive","title":"Improving Noise Robustness of Contrastive Speech Representation Learning with Speech Reconstruction","date":"2021-10-28","arxiv_id":"2110.15430","repositories_listed":0,"syntology":null},{"url":null,"slug":"closing-the-gap-between-time-domain-multi","title":"Closing the Gap Between Time-Domain Multi-Channel Speech Enhancement on Real and Simulation Conditions","date":"2021-10-27","arxiv_id":"2110.14139","repositories_listed":0,"syntology":null},{"url":null,"slug":"vida-man-visual-dialog-with-digital-humans","title":"ViDA-MAN: Visual Dialog with Digital Humans","date":"2021-10-26","arxiv_id":"2110.13384","repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-from-experiments-of-on-line-joint","title":"Findings from Experiments of On-line Joint Reinforcement Learning of Semantic Parser and Dialogue Manager with real Users","date":"2021-10-25","arxiv_id":"2110.13213","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-decentralized-distributed","title":"Asynchronous Decentralized Distributed Training of Acoustic Models","date":"2021-10-21","arxiv_id":"2110.11199","repositories_listed":0,"syntology":null},{"url":null,"slug":"synt-utilizing-imperfect-synthetic-data-to","title":"Synt++: Utilizing Imperfect Synthetic Data to Improve Speech Recognition","date":"2021-10-21","arxiv_id":"2110.11479","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-enhancing-ctc-model-for","title":"An Investigation of Enhancing CTC Model for Triggered Attention-based Streaming ASR","date":"2021-10-20","arxiv_id":"2110.10402","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-from-language-model-to","title":"Knowledge distillation from language model to acoustic model: a hierarchical multi-task learning approach","date":"2021-10-20","arxiv_id":"2110.10429","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-model-to-enhance-them-all-array-geometry","title":"One model to enhance them all: array geometry agnostic multi-channel personalized speech enhancement","date":"2021-10-20","arxiv_id":"2110.10330","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-domain-mapping-based-single-channel","title":"Progressive Learning for Stabilizing Label Selection in Speech Separation with Mapping-based Method","date":"2021-10-20","arxiv_id":"2110.10593","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-pattern-based-black-box-model","title":"Speech Pattern based Black-box Model Watermarking for Automatic Speech Recognition","date":"2021-10-19","arxiv_id":"2110.09814","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-learning-of-subword-dependent-model","title":"Automatic Learning of Subword Dependent Model Scales","date":"2021-10-18","arxiv_id":"2110.09324","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-sequence-training-of-attention","title":"Efficient Sequence Training of Attention Models using Approximative Recombination","date":"2021-10-18","arxiv_id":"2110.09245","repositories_listed":0,"syntology":null},{"url":null,"slug":"intent-classification-using-pre-trained","title":"Intent Classification Using Pre-trained Language Agnostic Embeddings For Low Resource Languages","date":"2021-10-18","arxiv_id":"2110.09264","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-speech-enhancement-new-models","title":"Personalized Speech Enhancement: New Models and Comprehensive Evaluation","date":"2021-10-18","arxiv_id":"2110.09625","repositories_listed":0,"syntology":null},{"url":null,"slug":"similarity-and-independence-aware-beamformer-1","title":"Similarity-and-Independence-Aware Beamformer with Iterative Casting and Boost Start for Target Source Extraction Using Reference","date":"2021-10-18","arxiv_id":"2110.09019","repositories_listed":0,"syntology":null},{"url":null,"slug":"virapart-a-text-refinement-framework-for-asr","title":"ViraPart: A Text Refinement Framework for Automatic Speech Recognition and Natural Language Processing Tasks in Persian","date":"2021-10-18","arxiv_id":"2110.09086","repositories_listed":0,"syntology":null},{"url":null,"slug":"okwugbe-end-to-end-speech-recognition-for-fon-1","title":"OkwuGbé: End-to-End Speech Recognition for Fon and Igbo","date":"2021-10-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-waveform-based-acoustic-models","title":"Towards Robust Waveform-Based Acoustic Models","date":"2021-10-16","arxiv_id":"2110.08634","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-and-challenges-in-deep-lip-reading","title":"Advances and Challenges in Deep Lip Reading","date":"2021-10-15","arxiv_id":"2110.07879","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-speech-recognition-using","title":"Multilingual Speech Recognition using Knowledge Transfer across Learning Processes","date":"2021-10-15","arxiv_id":"2110.07909","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-sparsity-dnn-fast-sparsity-optimization","title":"Omni-sparsity DNN: Fast Sparsity Optimization for On-Device Streaming E2E ASR via Supernet","date":"2021-10-15","arxiv_id":"2110.08352","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-identity-preserving-normal-to","title":"Towards Identity Preserving Normal to Dysarthric Voice Conversion","date":"2021-10-15","arxiv_id":"2110.08213","repositories_listed":0,"syntology":null},{"url":"/paper/sub-word-level-lip-reading-with-visual","slug":"sub-word-level-lip-reading-with-visual","title":"Sub-word Level Lip Reading With Visual Attention","date":"2021-10-14","arxiv_id":"2110.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"all-neural-beamformer-for-continuous-speech","title":"All-neural beamformer for continuous speech separation","date":"2021-10-13","arxiv_id":"2110.06428","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-using-lattice-free-mmi-for","title":"Continual learning using lattice-free MMI for speech recognition","date":"2021-10-13","arxiv_id":"2110.07055","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-domain-adaptation-of-language-1","title":"Prompt-tuning in ASR systems for efficient domain-adaptation","date":"2021-10-13","arxiv_id":"2110.06502","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-language-model-integration-for-rnn","title":"On Language Model Integration for RNN Transducer based Speech Recognition","date":"2021-10-13","arxiv_id":"2110.06841","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-point-identifying-critical","title":"Perception Point: Identifying Critical Learning Periods in Speech for Bilingual Networks","date":"2021-10-13","arxiv_id":"2110.06507","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-character-error-rate-is-not-equal","title":"Improving Character Error Rate Is Not Equal to Having Clean Speech: Speech Enhancement for ASR Systems with Black-box Acoustic Models","date":"2021-10-12","arxiv_id":"2110.05968","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-pre-training-for-automated-speech","title":"Multi-Modal Pre-Training for Automated Speech Recognition","date":"2021-10-12","arxiv_id":"2110.09890","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-summarization-using-restricted-self","title":"Speech Summarization using Restricted Self-Attention","date":"2021-10-12","arxiv_id":"2110.06263","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-order-does-not-matter-for-speech","title":"Word Order Does Not Matter For Speech Recognition","date":"2021-10-12","arxiv_id":"2110.05994","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-non-autoregressive","title":"A Comparative Study on Non-Autoregressive Modelings for Speech-to-Text Generation","date":"2021-10-11","arxiv_id":"2110.05249","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-momentum-pseudo-labeling-with","title":"Advancing Momentum Pseudo-Labeling with Conformer and Initialization Strategy","date":"2021-10-11","arxiv_id":"2110.04948","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-user-perception-of-speech","title":"Evaluating User Perception of Speech Recognition System Quality with Semantic Distance Metric","date":"2021-10-11","arxiv_id":"2110.05376","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-variable-training-for-efficient-on","title":"Partial Variable Training for Efficient On-Device Federated Learning","date":"2021-10-11","arxiv_id":"2110.05607","repositories_listed":0,"syntology":null},{"url":null,"slug":"sru-pioneering-fast-recurrence-with-attention","title":"SRU++: Pioneering Fast Recurrence with Attention for Speech Recognition","date":"2021-10-11","arxiv_id":"2110.05571","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2vec-switch-contrastive-learning-from","title":"Wav2vec-Switch: Contrastive Learning from Original-noisy Speech Pairs for Robust Speech Recognition","date":"2021-10-11","arxiv_id":"2110.04934","repositories_listed":0,"syntology":null},{"url":null,"slug":"have-best-of-both-worlds-two-pass-hybrid-and","title":"Have best of both worlds: two-pass hybrid and E2E cascading framework for speech recognition","date":"2021-10-10","arxiv_id":"2110.04891","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalizing-asr-with-limited-data-using","title":"DITTO: Data-efficient and Fair Targeted Subset Selection for ASR Accent Adaptation","date":"2021-10-10","arxiv_id":"2110.04908","repositories_listed":0,"syntology":null},{"url":"/paper/stepwise-refining-speech-separation-network","slug":"stepwise-refining-speech-separation-network","title":"Stepwise-Refining Speech Separation Network via Fine-Grained Encoding in High-order Latent Domain","date":"2021-10-10","arxiv_id":"2110.04791","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-high-fidelity-singing-voice","title":"Towards High-fidelity Singing Voice Conversion with Acoustic Reference and Contrastive Predictive Coding","date":"2021-10-10","arxiv_id":"2110.04754","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exploration-of-self-supervised-pretrained","title":"An Exploration of Self-Supervised Pretrained Representations for End-to-End Speech Recognition","date":"2021-10-09","arxiv_id":"2110.04590","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-with-locally-time-reversed","title":"Data Augmentation with Locally-time Reversed Speech for Automatic Speech Recognition","date":"2021-10-09","arxiv_id":"2110.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-automatic-speech-recognition","title":"Personalized Automatic Speech Recognition Trained on Small Disordered Speech Datasets","date":"2021-10-09","arxiv_id":"2110.04612","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2vec-s-semi-supervised-pre-training-for","title":"Wav2vec-S: Semi-Supervised Pre-Training for Low-Resource ASR","date":"2021-10-09","arxiv_id":"2110.04484","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-genetic-programming-approach-to-zero-shot","title":"A Genetic Programming Approach To Zero-Shot Neural Architecture Ranking","date":"2021-10-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-the-attention-mechanism-of-end-to","title":"Explaining the Attention Mechanism of End-to-End Speech Recognition Using Decision Trees","date":"2021-10-08","arxiv_id":"2110.03879","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-heterogeneous-characteristics-of","title":"Exploring Heterogeneous Characteristics of Layers in ASR Models for More Efficient Training","date":"2021-10-08","arxiv_id":"2110.04267","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-pseudo-label-training-for-end-to","title":"Improving Pseudo-label Training For End-to-end Speech Recognition Using Gradient Mask","date":"2021-10-08","arxiv_id":"2110.04056","repositories_listed":0,"syntology":null},{"url":null,"slug":"input-length-matters-an-empirical-study-of","title":"Input Length Matters: Improving RNN-T and MWER Training for Long-form Telephony Speech Recognition","date":"2021-10-08","arxiv_id":"2110.03841","repositories_listed":0,"syntology":null},{"url":null,"slug":"scala-supervised-contrastive-learning-for-end","title":"SCaLa: Supervised Contrastive Learning for End-to-End Speech Recognition","date":"2021-10-08","arxiv_id":"2110.04187","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-robust-automatic-speech-recognition","title":"Accent-Robust Automatic Speech Recognition Using Supervised and Unsupervised Wav2vec Embeddings","date":"2021-10-07","arxiv_id":"2110.03520","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-robustness-of-unsupervised","title":"Analyzing the Robustness of Unsupervised Speech Recognition","date":"2021-10-07","arxiv_id":"2110.03509","repositories_listed":0,"syntology":null},{"url":null,"slug":"back-from-the-future-bidirectional-ctc","title":"Back from the future: bidirectional CTC decoding using future information in speech recognition","date":"2021-10-07","arxiv_id":"2110.03326","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-on-device-training-of-speech","title":"Enabling On-Device Training of Speech Recognition Models with Federated Dropout","date":"2021-10-07","arxiv_id":"2110.03634","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-confidence-estimation-on-out-of","title":"Improving Confidence Estimation on Out-of-Domain Data for End-to-End Speech Recognition","date":"2021-10-07","arxiv_id":"2110.03327","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-neural-transducers","title":"Knowledge Distillation for Neural Transducers from Large Self-Supervised Pre-trained Models","date":"2021-10-07","arxiv_id":"2110.03334","repositories_listed":0,"syntology":null},{"url":null,"slug":"magic-dust-for-cross-lingual-adaptation-of","title":"Magic dust for cross-lingual adaptation of monolingual wav2vec-2.0","date":"2021-10-07","arxiv_id":"2110.03560","repositories_listed":0,"syntology":null},{"url":null,"slug":"mandarin-english-code-switching-speech","title":"Mandarin-English Code-switching Speech Recognition with Self-supervised Speech Representation Models","date":"2021-10-07","arxiv_id":"2110.03504","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-transformer-transducer-based-speech","title":"Streaming Transformer Transducer Based Speech Recognition Using Non-Causal Convolution","date":"2021-10-07","arxiv_id":"2110.05241","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribe-to-diarize-neural-speaker","title":"Transcribe-to-Diarize: Neural Speaker Diarization for Unlimited Number of Speakers using End-to-End Speaker-Attributed ASR","date":"2021-10-07","arxiv_id":"2110.03151","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-variations-through-new-wfst-topologies","title":"CTC Variations Through New WFST Topologies","date":"2021-10-06","arxiv_id":"2110.03098","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-categorical-features-in-end-to","title":"Integrating Categorical Features in End-to-End ASR","date":"2021-10-06","arxiv_id":"2110.03047","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-language-model-adaptation-with-text","title":"Internal Language Model Adaptation with Text-Only Data for End-to-End Speech Recognition","date":"2021-10-06","arxiv_id":"2110.05354","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-composition-of-weighted-finite-state","title":"Parallel Composition of Weighted Finite-State Transducers","date":"2021-10-06","arxiv_id":"2110.02848","repositories_listed":0,"syntology":null},{"url":null,"slug":"spell-my-name-keyword-boosted-speech","title":"Spell my name: keyword boosted speech recognition","date":"2021-10-06","arxiv_id":"2110.02791","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-rescoring-and-confidence-estimation-with","title":"ASR Rescoring and Confidence Estimation with ELECTRA","date":"2021-10-05","arxiv_id":"2110.01857","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-contextual-adaptation-with-neural","title":"Fast Contextual Adaptation with Neural Associative Memory for On-Device Personalized Speech Recognition","date":"2021-10-05","arxiv_id":"2110.02220","repositories_listed":0,"syntology":null},{"url":"/paper/is-attention-always-needed-a-case-study-on","slug":"is-attention-always-needed-a-case-study-on","title":"Is Attention always needed? A Case Study on Language Identification from Speech","date":"2021-10-05","arxiv_id":"2110.03427","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-noisy-audio-dataset-to-evaluate","title":"Building a Noisy Audio Dataset to Evaluate Machine Learning Approaches for Automatic Speech Recognition Systems","date":"2021-10-04","arxiv_id":"2110.01425","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-pre-trained-asr-models-for","title":"Exploiting Pre-Trained ASR Models for Alzheimer's Disease Recognition Through Spontaneous Speech","date":"2021-10-04","arxiv_id":"2110.01493","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-end-to-end-speech","title":"Towards efficient end-to-end speech recognition with biologically-inspired neural networks","date":"2021-10-04","arxiv_id":"2110.02743","repositories_listed":0,"syntology":null},{"url":"/paper/multi-task-voice-activated-framework-using","slug":"multi-task-voice-activated-framework-using","title":"Multi-task Voice Activated Framework using Self-supervised Learning","date":"2021-10-03","arxiv_id":"2110.01077","repositories_listed":0,"syntology":null},{"url":null,"slug":"significance-of-data-augmentation-for","title":"Significance of Data Augmentation for Improving Cleft Lip and Palate Speech Recognition","date":"2021-10-02","arxiv_id":"2110.00797","repositories_listed":0,"syntology":null},{"url":null,"slug":"chinese-medical-speech-recognition-with","title":"Chinese Medical Speech Recognition with Punctuated Hypothesis","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"data-centric-approach-to-chinese-medical","title":"Data centric approach to Chinese Medical Speech Recognition","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"2aba0e733de3f581c467d43be00d494a164b3f8ac8fc84d8dceceb7d264dc023","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}