{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/19","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":31,"rows_per_page":100,"rows":[1801,1900],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/18","next":"/task/automatic-speech-recognition/papers/20","papers":[{"url":null,"slug":"synt-utilizing-imperfect-synthetic-data-to","title":"Synt++: Utilizing Imperfect Synthetic Data to Improve Speech Recognition","date":"2021-10-21","arxiv_id":"2110.11479","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-enhancing-ctc-model-for","title":"An Investigation of Enhancing CTC Model for Triggered Attention-based Streaming ASR","date":"2021-10-20","arxiv_id":"2110.10402","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-model-to-enhance-them-all-array-geometry","title":"One model to enhance them all: array geometry agnostic multi-channel personalized speech enhancement","date":"2021-10-20","arxiv_id":"2110.10330","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-pattern-based-black-box-model","title":"Speech Pattern based Black-box Model Watermarking for Automatic Speech Recognition","date":"2021-10-19","arxiv_id":"2110.09814","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-learning-of-subword-dependent-model","title":"Automatic Learning of Subword Dependent Model Scales","date":"2021-10-18","arxiv_id":"2110.09324","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-sequence-training-of-attention","title":"Efficient Sequence Training of Attention Models using Approximative Recombination","date":"2021-10-18","arxiv_id":"2110.09245","repositories_listed":0,"syntology":null},{"url":null,"slug":"intent-classification-using-pre-trained","title":"Intent Classification Using Pre-trained Language Agnostic Embeddings For Low Resource Languages","date":"2021-10-18","arxiv_id":"2110.09264","repositories_listed":0,"syntology":null},{"url":null,"slug":"virapart-a-text-refinement-framework-for-asr","title":"ViraPart: A Text Refinement Framework for Automatic Speech Recognition and Natural Language Processing Tasks in Persian","date":"2021-10-18","arxiv_id":"2110.09086","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-speech-recognition-using","title":"Multilingual Speech Recognition using Knowledge Transfer across Learning Processes","date":"2021-10-15","arxiv_id":"2110.07909","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-sparsity-dnn-fast-sparsity-optimization","title":"Omni-sparsity DNN: Fast Sparsity Optimization for On-Device Streaming E2E ASR via Supernet","date":"2021-10-15","arxiv_id":"2110.08352","repositories_listed":0,"syntology":null},{"url":"/paper/sub-word-level-lip-reading-with-visual","slug":"sub-word-level-lip-reading-with-visual","title":"Sub-word Level Lip Reading With Visual Attention","date":"2021-10-14","arxiv_id":"2110.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"all-neural-beamformer-for-continuous-speech","title":"All-neural beamformer for continuous speech separation","date":"2021-10-13","arxiv_id":"2110.06428","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-using-lattice-free-mmi-for","title":"Continual learning using lattice-free MMI for speech recognition","date":"2021-10-13","arxiv_id":"2110.07055","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-domain-adaptation-of-language-1","title":"Prompt-tuning in ASR systems for efficient domain-adaptation","date":"2021-10-13","arxiv_id":"2110.06502","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-character-error-rate-is-not-equal","title":"Improving Character Error Rate Is Not Equal to Having Clean Speech: Speech Enhancement for ASR Systems with Black-box Acoustic Models","date":"2021-10-12","arxiv_id":"2110.05968","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-order-does-not-matter-for-speech","title":"Word Order Does Not Matter For Speech Recognition","date":"2021-10-12","arxiv_id":"2110.05994","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-non-autoregressive","title":"A Comparative Study on Non-Autoregressive Modelings for Speech-to-Text Generation","date":"2021-10-11","arxiv_id":"2110.05249","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-momentum-pseudo-labeling-with","title":"Advancing Momentum Pseudo-Labeling with Conformer and Initialization Strategy","date":"2021-10-11","arxiv_id":"2110.04948","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-user-perception-of-speech","title":"Evaluating User Perception of Speech Recognition System Quality with Semantic Distance Metric","date":"2021-10-11","arxiv_id":"2110.05376","repositories_listed":0,"syntology":null},{"url":null,"slug":"sru-pioneering-fast-recurrence-with-attention","title":"SRU++: Pioneering Fast Recurrence with Attention for Speech Recognition","date":"2021-10-11","arxiv_id":"2110.05571","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2vec-switch-contrastive-learning-from","title":"Wav2vec-Switch: Contrastive Learning from Original-noisy Speech Pairs for Robust Speech Recognition","date":"2021-10-11","arxiv_id":"2110.04934","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalizing-asr-with-limited-data-using","title":"DITTO: Data-efficient and Fair Targeted Subset Selection for ASR Accent Adaptation","date":"2021-10-10","arxiv_id":"2110.04908","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-high-fidelity-singing-voice","title":"Towards High-fidelity Singing Voice Conversion with Acoustic Reference and Contrastive Predictive Coding","date":"2021-10-10","arxiv_id":"2110.04754","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exploration-of-self-supervised-pretrained","title":"An Exploration of Self-Supervised Pretrained Representations for End-to-End Speech Recognition","date":"2021-10-09","arxiv_id":"2110.04590","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-with-locally-time-reversed","title":"Data Augmentation with Locally-time Reversed Speech for Automatic Speech Recognition","date":"2021-10-09","arxiv_id":"2110.04511","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-automatic-speech-recognition","title":"Personalized Automatic Speech Recognition Trained on Small Disordered Speech Datasets","date":"2021-10-09","arxiv_id":"2110.04612","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2vec-s-semi-supervised-pre-training-for","title":"Wav2vec-S: Semi-Supervised Pre-Training for Low-Resource ASR","date":"2021-10-09","arxiv_id":"2110.04484","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-genetic-programming-approach-to-zero-shot","title":"A Genetic Programming Approach To Zero-Shot Neural Architecture Ranking","date":"2021-10-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-heterogeneous-characteristics-of","title":"Exploring Heterogeneous Characteristics of Layers in ASR Models for More Efficient Training","date":"2021-10-08","arxiv_id":"2110.04267","repositories_listed":0,"syntology":null},{"url":null,"slug":"input-length-matters-an-empirical-study-of","title":"Input Length Matters: Improving RNN-T and MWER Training for Long-form Telephony Speech Recognition","date":"2021-10-08","arxiv_id":"2110.03841","repositories_listed":0,"syntology":null},{"url":null,"slug":"scala-supervised-contrastive-learning-for-end","title":"SCaLa: Supervised Contrastive Learning for End-to-End Speech Recognition","date":"2021-10-08","arxiv_id":"2110.04187","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-robust-automatic-speech-recognition","title":"Accent-Robust Automatic Speech Recognition Using Supervised and Unsupervised Wav2vec Embeddings","date":"2021-10-07","arxiv_id":"2110.03520","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-on-device-training-of-speech","title":"Enabling On-Device Training of Speech Recognition Models with Federated Dropout","date":"2021-10-07","arxiv_id":"2110.03634","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-confidence-estimation-on-out-of","title":"Improving Confidence Estimation on Out-of-Domain Data for End-to-End Speech Recognition","date":"2021-10-07","arxiv_id":"2110.03327","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-neural-transducers","title":"Knowledge Distillation for Neural Transducers from Large Self-Supervised Pre-trained Models","date":"2021-10-07","arxiv_id":"2110.03334","repositories_listed":0,"syntology":null},{"url":null,"slug":"magic-dust-for-cross-lingual-adaptation-of","title":"Magic dust for cross-lingual adaptation of monolingual wav2vec-2.0","date":"2021-10-07","arxiv_id":"2110.03560","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribe-to-diarize-neural-speaker","title":"Transcribe-to-Diarize: Neural Speaker Diarization for Unlimited Number of Speakers using End-to-End Speaker-Attributed ASR","date":"2021-10-07","arxiv_id":"2110.03151","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-variations-through-new-wfst-topologies","title":"CTC Variations Through New WFST Topologies","date":"2021-10-06","arxiv_id":"2110.03098","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-language-model-adaptation-with-text","title":"Internal Language Model Adaptation with Text-Only Data for End-to-End Speech Recognition","date":"2021-10-06","arxiv_id":"2110.05354","repositories_listed":0,"syntology":null},{"url":null,"slug":"spell-my-name-keyword-boosted-speech","title":"Spell my name: keyword boosted speech recognition","date":"2021-10-06","arxiv_id":"2110.02791","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-rescoring-and-confidence-estimation-with","title":"ASR Rescoring and Confidence Estimation with ELECTRA","date":"2021-10-05","arxiv_id":"2110.01857","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-contextual-adaptation-with-neural","title":"Fast Contextual Adaptation with Neural Associative Memory for On-Device Personalized Speech Recognition","date":"2021-10-05","arxiv_id":"2110.02220","repositories_listed":0,"syntology":null},{"url":"/paper/is-attention-always-needed-a-case-study-on","slug":"is-attention-always-needed-a-case-study-on","title":"Is Attention always needed? A Case Study on Language Identification from Speech","date":"2021-10-05","arxiv_id":"2110.03427","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-noisy-audio-dataset-to-evaluate","title":"Building a Noisy Audio Dataset to Evaluate Machine Learning Approaches for Automatic Speech Recognition Systems","date":"2021-10-04","arxiv_id":"2110.01425","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-pre-trained-asr-models-for","title":"Exploiting Pre-Trained ASR Models for Alzheimer's Disease Recognition Through Spontaneous Speech","date":"2021-10-04","arxiv_id":"2110.01493","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-end-to-end-speech","title":"Towards efficient end-to-end speech recognition with biologically-inspired neural networks","date":"2021-10-04","arxiv_id":"2110.02743","repositories_listed":0,"syntology":null},{"url":null,"slug":"chinese-medical-speech-recognition-with","title":"Chinese Medical Speech Recognition with Punctuated Hypothesis","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"employing-low-pass-filtered-temporal-speech","title":"Employing low-pass filtered temporal speech features for the training of ideal ratio mask in speech enhancement","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-integration-of-e2e-asr-and","title":"Exploring the Integration of E2E ASR and Pronunciation Modeling for English Mispronunciation Detection","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-technology-for-everyone-automatic","title":"Speech Technology for Everyone: Automatic Speech Recognition for Non-Native English with Transfer Learning","date":"2021-10-01","arxiv_id":"2110.00678","repositories_listed":0,"syntology":null},{"url":null,"slug":"spliceout-a-simple-and-efficient-audio","title":"SpliceOut: A Simple and Efficient Audio Augmentation Method","date":"2021-09-30","arxiv_id":"2110.00046","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-self-supervised-speech-pre","title":"Comparison of Self-Supervised Speech Pre-Training Methods on Flemish Dutch","date":"2021-09-29","arxiv_id":"2109.14357","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-sequence-to-sequence-networks","title":"Conditioning Sequence-to-sequence Networks with Learned Activations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-limited-adversarial","title":"Demystifying Limited Adversarial Transferability in Automatic Speech Recognition Systems","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mlp-based-architecture-with-variable-length","title":"MLP-based architecture with variable length input for automatic speech recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"phasefool-phase-oriented-audio-adversarial","title":"PhaseFool: Phase-oriented Audio Adversarial Examples via Energy Dissipation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesising-audio-adversarial-examples-for","title":"Synthesising Audio Adversarial Examples for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-role-of-self-attention-for","title":"Understanding the Role of Self Attention for Efficient Speech Recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"w-ctc-a-connectionist-temporal-classification","title":"W-CTC: a Connectionist Temporal Classification Loss with Wild Cards","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"private-language-model-adaptation-for-speech","title":"Private Language Model Adaptation for Speech Recognition","date":"2021-09-28","arxiv_id":"2110.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-level-confidence-estimation-for-rnn","title":"Word-level confidence estimation for RNN transducers","date":"2021-09-28","arxiv_id":"2110.15222","repositories_listed":0,"syntology":null},{"url":"/paper/bigssl-exploring-the-frontier-of-large-scale","slug":"bigssl-exploring-the-frontier-of-large-scale","title":"BigSSL: Exploring the Frontier of Large-Scale Semi-Supervised Learning for Automatic Speech Recognition","date":"2021-09-27","arxiv_id":"2109.13226","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-and-opportunities-of-speech","title":"Challenges and Opportunities of Speech Recognition for Bengali Language","date":"2021-09-27","arxiv_id":"2109.13217","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-model-robustness-to-automatic-speech","title":"Topic Model Robustness to Automatic Speech Recognition Errors in Podcast Transcripts","date":"2021-09-25","arxiv_id":"2109.12306","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-domain-specific-language-models-for","title":"Learning Domain Specific Language Models for Automatic Speech Recognition through Machine Translation","date":"2021-09-21","arxiv_id":"2110.10261","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-recognition-is-worth-32","title":"Audio-Visual Speech Recognition is Worth 32$\\times$32$\\times$8 Voxels","date":"2021-09-20","arxiv_id":"2109.09536","repositories_listed":0,"syntology":null},{"url":null,"slug":"irnn-integer-only-recurrent-neural-network","title":"iRNN: Integer-only Recurrent Neural Network","date":"2021-09-20","arxiv_id":"2109.09828","repositories_listed":0,"syntology":null},{"url":null,"slug":"meetdot-videoconferencing-with-live","title":"MeetDot: Videoconferencing with Live Translation Captions","date":"2021-09-20","arxiv_id":"2109.09577","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-approach-for-measuring-the","title":"Model-Based Approach for Measuring the Fairness in ASR","date":"2021-09-19","arxiv_id":"2109.09061","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-audio-textual-architecture-for","title":"Multimodal Audio-textual Architecture for Robust Spoken Language Understanding","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pdaugment-data-augmentation-by-pitch-and","title":"PDAugment: Data Augmentation by Pitch and Duration Adjustments for Automatic Lyrics Transcription","date":"2021-09-16","arxiv_id":"2109.07940","repositories_listed":0,"syntology":null},{"url":null,"slug":"utterance-level-neural-confidence-measure-for","title":"Utterance-level neural confidence measure for end-to-end children speech recognition","date":"2021-09-16","arxiv_id":"2109.07750","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accent-identification-and-accented","title":"Improving Accent Identification and Accented Speech Recognition Under a Framework of Self-supervised Learning","date":"2021-09-15","arxiv_id":"2109.07349","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-streaming-transformer-based-asr","title":"Improving Streaming Transformer Based ASR Under a Framework of Self-supervised Learning","date":"2021-09-15","arxiv_id":"2109.07327","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-transformer-with-unified","title":"Non-autoregressive Transformer with Unified Bidirectional Decoder for Automatic Speech Recognition","date":"2021-09-14","arxiv_id":"2109.06684","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-adapters-for-parameter-efficient-asr","title":"Residual Adapters for Parameter-Efficient ASR Adaptation to Atypical and Accented Speech","date":"2021-09-14","arxiv_id":"2109.06952","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-schemes-for","title":"Unsupervised Domain Adaptation Schemes for Building ASR in Low-resource Languages","date":"2021-09-12","arxiv_id":"2109.05494","repositories_listed":0,"syntology":null},{"url":null,"slug":"remember-the-context-asr-slot-error","title":"Remember the context! ASR slot error correction through memorization","date":"2021-09-10","arxiv_id":"2109.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-attention-channel-combinator-frontend","title":"Self-Attention Channel Combinator Frontend for End-to-End Multichannel Far-field Speech Recognition","date":"2021-09-10","arxiv_id":"2109.04783","repositories_listed":0,"syntology":null},{"url":null,"slug":"coarse-to-fine-and-cross-lingual-asr-transfer","title":"Coarse-To-Fine And Cross-Lingual ASR Transfer","date":"2021-09-02","arxiv_id":"2109.00916","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-end-to-end-automatic-speech-1","title":"Robustness of end-to-end Automatic Speech Recognition Models – A Case Study using Mozilla DeepSpeech","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-constrained-pointer-generator-for-end-to","title":"Tree-constrained Pointer Generator for End-to-end Contextual Speech Recognition","date":"2021-09-01","arxiv_id":"2109.00627","repositories_listed":0,"syntology":null},{"url":"/paper/asr-glue-a-new-multi-task-benchmark-for-asr","slug":"asr-glue-a-new-multi-task-benchmark-for-asr","title":"ASR-GLUE: A New Multi-task Benchmark for ASR-Robust Natural Language Understanding","date":"2021-08-30","arxiv_id":"2108.13048","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigations-on-speech-recognition-systems","title":"Investigations on Speech Recognition Systems for Low-Resource Dialectal Arabic-English Code-Switching Speech","date":"2021-08-29","arxiv_id":"2108.12881","repositories_listed":0,"syntology":null},{"url":null,"slug":"4-bit-quantization-of-lstm-based-speech","title":"4-bit Quantization of LSTM-based Speech Recognition Models","date":"2021-08-27","arxiv_id":"2108.12074","repositories_listed":0,"syntology":null},{"url":null,"slug":"grammar-based-identification-of-speaker-role","title":"Grammar Based Speaker Role Identification for Air Traffic Control Speech Recognition","date":"2021-08-27","arxiv_id":"2108.12175","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-callsign-recognition-with-air","title":"Improving callsign recognition with air-surveillance data in air-traffic communication","date":"2021-08-27","arxiv_id":"2108.12156","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-aware-warping-factors-in-mask-based","title":"Task-aware Warping Factors in Mask-based Speech Enhancement","date":"2021-08-27","arxiv_id":"2108.12128","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-single-channel-speech","title":"Cross-domain Single-channel Speech Enhancement Model with Bi-projection Fusion Module for Noise-robust ASR","date":"2021-08-26","arxiv_id":"2108.11598","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-exposure-bias-in-training-recurrent","title":"Reducing Exposure Bias in Training Recurrent Neural Network Transducers","date":"2021-08-24","arxiv_id":"2108.10803","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-transformer-based-framework-for","title":"A Unified Transformer-based Framework for Duplex Text Normalization","date":"2021-08-23","arxiv_id":"2108.09889","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-using-limited","title":"Automatic Speech Recognition And Limited Vocabulary: A Survey","date":"2021-08-23","arxiv_id":"2108.10254","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-summarization-for-longform","title":"Hierarchical Summarization for Longform Spoken Dialog","date":"2021-08-21","arxiv_id":"2108.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-streams-and-two-resolution-spectrograms","title":"A Multi-level Acoustic Feature Extraction Framework for Transformer Based End-to-End Speech Recognition","date":"2021-08-18","arxiv_id":"2108.07980","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-light-weight-contextual-spelling-correction","title":"A Light-weight contextual spelling correction model for customizing transducer-based speech recognition systems","date":"2021-08-17","arxiv_id":"2108.07493","repositories_listed":0,"syntology":null},{"url":null,"slug":"stargan-vc-asr-stargan-based-non-parallel","title":"StarGAN-VC+ASR: StarGAN-based Non-Parallel Voice Conversion Regularized by Automatic Speech Recognition","date":"2021-08-10","arxiv_id":"2108.04395","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hw-tsc-s-offline-speech-translation","title":"The HW-TSC's Offline Speech Translation Systems for IWSLT 2021 Evaluation","date":"2021-08-09","arxiv_id":"2108.03845","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-and-neural-network-guided-convolutional","title":"Blind and neural network-guided convolutional beamformer for joint denoising, dereverberation, and source separation","date":"2021-08-04","arxiv_id":"2108.01836","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyn-asr-compact-multilingual-speech","title":"Dyn-ASR: Compact, Multilingual Speech Recognition via Spoken Language and Accent Identification","date":"2021-08-04","arxiv_id":"2108.02034","repositories_listed":0,"syntology":null}],"record_sha256":"4080c26d3fe36fb2dce4d6447e2ca7272384d092e16b8b4fbc1d138e20af11cf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}