{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/21","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":21,"pages_in_order":31,"rows_per_page":100,"rows":[2001,2100],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/20","next":"/task/automatic-speech-recognition/papers/22","papers":[{"url":null,"slug":"english-accent-accuracy-analysis-in-a-state","title":"English Accent Accuracy Analysis in a State-of-the-Art Automatic Speech Recognition System","date":"2021-05-09","arxiv_id":"2105.05041","repositories_listed":0,"syntology":null},{"url":null,"slug":"latency-controlled-neural-architecture-search","title":"Latency-Controlled Neural Architecture Search for Streaming Speech Recognition","date":"2021-05-08","arxiv_id":"2105.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-end-to-end-automatic-speech","title":"Robustness of end-to-end Automatic Speech Recognition Models -- A Case Study using Mozilla DeepSpeech","date":"2021-05-08","arxiv_id":"2105.09742","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-recognition-with-hybrid-phonetic","title":"Accent Recognition with Hybrid Phonetic Features","date":"2021-05-05","arxiv_id":"2105.01920","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-modification-for-recognition-of","title":"Spectral modification for recognition of children’s speech undermismatched conditions","date":"2021-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-keyphrase-detection-using","title":"Personalized Keyphrase Detection using Speaker and Environment Information","date":"2021-04-28","arxiv_id":"2104.13970","repositories_listed":0,"syntology":null},{"url":null,"slug":"head-synchronous-decoding-for-transformer","title":"Head-synchronous Decoding for Transformer-based Streaming ASR","date":"2021-04-26","arxiv_id":"2104.12631","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-for-end-to-end-asr-word","title":"Multi-Task Learning for End-to-End ASR Word and Utterance Confidence with Deletion Prediction","date":"2021-04-26","arxiv_id":"2104.12870","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-data-augmentation-for-end-to-end","title":"Semantic Data Augmentation for End-to-End Mandarin Speech Recognition","date":"2021-04-26","arxiv_id":"2104.12521","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-streaming-and-non","title":"Bridging the gap between streaming and non-streaming ASR systems bydistilling ensembles of CTC and RNN-T models","date":"2021-04-25","arxiv_id":"2104.14346","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-of-deep-neural-networks-for-1","title":"Quantization of Deep Neural Networks for Accurate Edge Computing","date":"2021-04-25","arxiv_id":"2104.12046","repositories_listed":0,"syntology":null},{"url":null,"slug":"accented-speech-recognition-a-survey","title":"Accented Speech Recognition: A Survey","date":"2021-04-21","arxiv_id":"2104.10747","repositories_listed":0,"syntology":null},{"url":null,"slug":"discriminative-self-training-for-punctuation","title":"Discriminative Self-training for Punctuation Prediction","date":"2021-04-21","arxiv_id":"2104.10339","repositories_listed":0,"syntology":null},{"url":null,"slug":"disfluency-detection-with-unlabeled-data-and","title":"Disfluency Detection with Unlabeled Data and Small BERT Models","date":"2021-04-21","arxiv_id":"2104.10769","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-synchronous-speech-to-text-alignment","title":"Label-Synchronous Speech-to-Text Alignment for ASR Using Forward and Backward Transformers","date":"2021-04-21","arxiv_id":"2104.10328","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-sampling-based-training-criteria-for","title":"On Sampling-Based Training Criteria for Neural Language Modeling","date":"2021-04-21","arxiv_id":"2104.10507","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-for-spoken-language","title":"Pre-training for Spoken Language Understanding with Joint Textual and Phonetic Representation Learning","date":"2021-04-21","arxiv_id":"2104.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-aware-far-field-automatic-speech","title":"Scene-aware Far-field Automatic Speech Recognition","date":"2021-04-21","arxiv_id":"2104.10757","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-impact-of-word-error-rate-on-acoustic","title":"On the Impact of Word Error Rate on Acoustic-Linguistic Speech Emotion Recognition: An Update for the Deep Learning Era","date":"2021-04-20","arxiv_id":"2104.10121","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-data-driven-subword-modeling-for-end","title":"Acoustic Data-Driven Subword Modeling for End-to-End Speech Recognition","date":"2021-04-19","arxiv_id":"2104.09106","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-long-context-end-to-end-speech","title":"Advanced Long-context End-to-end Speech Recognition Using Context-expanded Transformers","date":"2021-04-19","arxiv_id":"2104.09426","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimo-self-attentive-rnn-beamformer-for-multi","title":"MIMO Self-attentive RNN Beamformer for Multi-speaker Speech Separation","date":"2021-04-17","arxiv_id":"2104.08450","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-clean-data-training","title":"Bridging the Gap Between Clean Data Training and Real-World Inference for Spoken Language Understanding","date":"2021-04-13","arxiv_id":"2104.06393","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivalence-of-segmental-and-neural","title":"Equivalence of Segmental and Neural Transducer Modeling: A Proof of Concept","date":"2021-04-13","arxiv_id":"2104.06104","repositories_listed":0,"syntology":null},{"url":null,"slug":"source-and-target-bidirectional-knowledge","title":"Source and Target Bidirectional Knowledge Distillation for End-to-end Speech Translation","date":"2021-04-13","arxiv_id":"2104.06457","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-the-benefit-of-synthetic-training","title":"Comparing the Benefit of Synthetic Training Data for Various Automatic Speech Recognition Architectures","date":"2021-04-12","arxiv_id":"2104.05379","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-conformer-based-end-to-end-speech","title":"Improved Conformer-based End-to-End Speech Recognition Using Neural Architecture Search","date":"2021-04-12","arxiv_id":"2104.05390","repositories_listed":0,"syntology":null},{"url":null,"slug":"innovative-bert-based-reranking-language","title":"Innovative Bert-based Reranking Language Models for Speech Recognition","date":"2021-04-11","arxiv_id":"2104.04950","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-transformer-based-end-to","title":"Non-autoregressive Transformer-based End-to-end ASR using BERT","date":"2021-04-10","arxiv_id":"2104.04805","repositories_listed":0,"syntology":null},{"url":null,"slug":"accented-speech-recognition-inspired-by-human","title":"Accented Speech Recognition Inspired by Human Perception","date":"2021-04-09","arxiv_id":"2104.04627","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-replacement-and-combination-for","title":"On Architectures and Training for Raw Waveform Feature Extraction in ASR","date":"2021-04-09","arxiv_id":"2104.04298","repositories_listed":0,"syntology":null},{"url":"/paper/bstc-a-large-scale-chinese-english-speech","slug":"bstc-a-large-scale-chinese-english-speech","title":"BSTC: A Large-Scale Chinese-English Speech Translation Dataset","date":"2021-04-08","arxiv_id":"2104.03575","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-semi-supervised-learning-an","title":"Contextual Semi-Supervised Learning: An Approach To Leverage Air-Surveillance and Untranscribed ATC Data in ASR Systems","date":"2021-04-08","arxiv_id":"2104.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-machine-speech-chain-for-domain","title":"Exploring Machine Speech Chain for Domain Adaptation and Few-Shot Speaker Adaptation","date":"2021-04-08","arxiv_id":"2104.03815","repositories_listed":0,"syntology":null},{"url":null,"slug":"wnars-wfst-based-non-autoregressive-streaming","title":"WNARS: WFST based Non-autoregressive Streaming End-to-End Speech Recognition","date":"2021-04-08","arxiv_id":"2104.03587","repositories_listed":0,"syntology":null},{"url":null,"slug":"capturing-multi-resolution-context-by-dilated","title":"Capturing Multi-Resolution Context by Dilated Self-Attention","date":"2021-04-07","arxiv_id":"2104.02858","repositories_listed":0,"syntology":null},{"url":null,"slug":"pushing-the-limits-of-non-autoregressive","title":"Pushing the Limits of Non-Autoregressive Speech Recognition","date":"2021-04-07","arxiv_id":"2104.03416","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-ctc-and-lfmmi-for-out-of-domain","title":"Comparing CTC and LFMMI for out-of-domain adaptation of wav2vec 2.0 acoustic model","date":"2021-04-06","arxiv_id":"2104.02558","repositories_listed":0,"syntology":null},{"url":null,"slug":"dissecting-user-perceived-latency-of-on","title":"Dissecting User-Perceived Latency of On-Device E2E Speech Recognition","date":"2021-04-06","arxiv_id":"2104.02207","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-targeted-universal-adversarial","title":"Exploring Targeted Universal Adversarial Perturbations to End-to-end ASR Models","date":"2021-04-06","arxiv_id":"2104.02757","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexi-transducer-optimizing-latency-accuracy","title":"Flexi-Transducer: Optimizing Latency, Accuracy and Compute forMulti-Domain On-Device Scenarios","date":"2021-04-06","arxiv_id":"2104.02232","repositories_listed":0,"syntology":null},{"url":null,"slug":"relaxing-the-conditional-independence","title":"Relaxing the Conditional Independence Assumption of CTC-based ASR by Conditioning on Intermediate Predictions","date":"2021-04-06","arxiv_id":"2104.02724","repositories_listed":0,"syntology":null},{"url":null,"slug":"citrinet-closing-the-gap-between-non","title":"Citrinet: Closing the Gap between Non-Autoregressive and Autoregressive End-to-End Models for Automatic Speech Recognition","date":"2021-04-05","arxiv_id":"2104.01721","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speaker-attributed-asr-with","title":"End-to-End Speaker-Attributed ASR with Transformer","date":"2021-04-05","arxiv_id":"2104.02128","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-distance-a-new-metric-for-asr","title":"Semantic Distance: A New Metric for ASR Performance Analysis Towards Spoken Language Understanding","date":"2021-04-05","arxiv_id":"2104.02138","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-diarization-assisted-asr-for-multi","title":"Speaker conditioned acoustic modeling for multi-speaker conversational ASR","date":"2021-04-05","arxiv_id":"2104.01882","repositories_listed":0,"syntology":null},{"url":null,"slug":"talk-don-t-write-a-study-of-direct-speech","title":"Talk, Don't Write: A Study of Direct Speech-Based Image Retrieval","date":"2021-04-05","arxiv_id":"2104.01894","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-lifelong-learning-of-end-to-end-asr","title":"Towards Lifelong Learning of End-to-end ASR","date":"2021-04-04","arxiv_id":"2104.01616","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-joint-training-with-self","title":"Adversarial Joint Training with Self-Attention Mechanism for Robust End-to-End Speech Recognition","date":"2021-04-03","arxiv_id":"2104.01471","repositories_listed":0,"syntology":null},{"url":null,"slug":"configurable-privacy-preserving-automatic","title":"Configurable Privacy-Preserving Automatic Speech Recognition","date":"2021-04-01","arxiv_id":"2104.00766","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-sensitive-evaluation-of-automatic","title":"Context-sensitive evaluation of automatic speech recognition: considering user experience & language variation","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dialect-identification-through-adversarial","title":"Dialect Identification through Adversarial Learning and Knowledge Distillation on Romanian BERT","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-end-to-end-asr-for-endangered-1","title":"Leveraging End-to-End ASR for Endangered Language Documentation: An Empirical Study on Yol\\'oxochitl Mixtec","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tutorial-proposal-end-to-end-speech","title":"Tutorial Proposal: End-to-End Speech Translation","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-and-defenses-for-speech","title":"Adversarial Attacks and Defenses for Speech Recognition Systems","date":"2021-03-31","arxiv_id":"2103.17122","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-pre-training-of-end-to-end-multi","title":"Large-Scale Pre-Training of End-to-End Multi-Talker ASR for Meeting Transcription with Single Distant Microphone","date":"2021-03-31","arxiv_id":"2103.16776","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-encoder-learning-and-stream-fusion-for","title":"Multi-Encoder Learning and Stream Fusion for Transformer-Based End-to-End Automatic Speech Recognition","date":"2021-03-31","arxiv_id":"2104.00120","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-hypothesis-ctc-based-semi-supervised","title":"Multiple-hypothesis CTC-based semi-supervised adaptation of end-to-end speech recognition","date":"2021-03-29","arxiv_id":"2103.15515","repositories_listed":0,"syntology":null},{"url":null,"slug":"bart-based-semantic-correction-for-mandarin","title":"BART based semantic correction for Mandarin automatic speech recognition system","date":"2021-03-26","arxiv_id":"2104.05507","repositories_listed":0,"syntology":null},{"url":"/paper/construction-of-a-large-scale-japanese-asr","slug":"construction-of-a-large-scale-japanese-asr","title":"Construction of a Large-scale Japanese ASR Corpus on TV Recordings","date":"2021-03-26","arxiv_id":"2103.14736","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutually-constrained-monotonic-multihead","title":"Mutually-Constrained Monotonic Multihead Attention for Online ASR","date":"2021-03-26","arxiv_id":"2103.14302","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-approach-to-improve-robustness-of-nlp","title":"An Approach to Improve Robustness of NLP Systems against ASR Errors","date":"2021-03-25","arxiv_id":"2103.13610","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-energy-based-models-for-end-to-end","title":"Residual Energy-Based Models for End-to-End Speech Recognition","date":"2021-03-25","arxiv_id":"2103.14152","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-privacy-with-smart-digital-assistants","title":"Voice Privacy with Smart Digital Assistants in Educational Settings","date":"2021-03-24","arxiv_id":"2104.11038","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-of-speech-recognition-errors","title":"Hallucination of speech recognition errors with sequence to sequence learning","date":"2021-03-23","arxiv_id":"2103.12258","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-biasing-of-language-models-for","title":"Contextual Biasing of Language Models for Speech Recognition in Goal-Oriented Conversational Agents","date":"2021-03-18","arxiv_id":"2103.10325","repositories_listed":0,"syntology":null},{"url":"/paper/transformer-based-asr-incorporating-time","slug":"transformer-based-asr-incorporating-time","title":"Transformer-based ASR Incorporating Time-reduction Layer and Fine-tuning with Self-Knowledge Distillation","date":"2021-03-17","arxiv_id":"2103.09903","repositories_listed":0,"syntology":null},{"url":"/paper/edgecrnn-an-edgecomputing-oriented-model-of","slug":"edgecrnn-an-edgecomputing-oriented-model-of","title":"EdgeCRNN: an edgecomputing oriented model of acoustic feature enhancement for keyword spotting","date":"2021-03-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributed-optimisation-framework","title":"A Distributed Optimisation Framework Combining Natural Gradient with Hessian-Free for Discriminative Sequence Training","date":"2021-03-12","arxiv_id":"2103.07554","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-acoustic-unit-augmentation-with-bpe","title":"Dynamic Acoustic Unit Augmentation With BPE-Dropout for Low-Resource End-to-End Speech Recognition","date":"2021-03-12","arxiv_id":"2103.07186","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-word-level-confidence-for-subword","title":"Learning Word-Level Confidence For Subword End-to-End ASR","date":"2021-03-11","arxiv_id":"2103.06716","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-of-both-worlds-robust-accented-speech","title":"Best of Both Worlds: Robust Accented Speech Recognition with Adversarial Transfer Learning","date":"2021-03-10","arxiv_id":"2103.05834","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-semi-supervised-learning-for-asr","title":"Contrastive Semi-supervised Learning for ASR","date":"2021-03-09","arxiv_id":"2103.05149","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ultra-low-power-rnn-classifier-for-always","title":"An Ultra-low Power RNN Classifier for Always-On Voice Wake-Up Detection Robust to Real-World Scenarios","date":"2021-03-08","arxiv_id":"2103.04792","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-model-robustness-for-skill-routing-in","title":"Neural model robustness for skill routing in large-scale conversational AI systems: A design choice exploration","date":"2021-03-04","arxiv_id":"2103.03373","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-running-speech-recognizer-an-end-to-end","title":"Incorporating VAD into ASR System by Multi-task Learning","date":"2021-03-02","arxiv_id":"2103.01661","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-knowledge-distillation-for-online","title":"Alignment Knowledge Distillation for Online Streaming Attention-based Speech Recognition","date":"2021-02-28","arxiv_id":"2103.00422","repositories_listed":0,"syntology":null},{"url":null,"slug":"brain-signals-to-rescue-aphasia-apraxia-and","title":"Brain Signals to Rescue Aphasia, Apraxia and Dysarthria Speech Recognition","date":"2021-02-28","arxiv_id":"2103.00383","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-for-improving-rare-word","title":"Meta-Learning for improving rare word recognition in end-to-end ASR","date":"2021-02-25","arxiv_id":"2102.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixspeech-data-augmentation-for-low-resource","title":"MixSpeech: Data Augmentation for Low-resource Automatic Speech Recognition","date":"2021-02-25","arxiv_id":"2102.12664","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-using-multi-stage-self","title":"Speech Enhancement Using Multi-Stage Self-Attentive Temporal Convolutional Networks","date":"2021-02-24","arxiv_id":"2102.12078","repositories_listed":0,"syntology":null},{"url":null,"slug":"thoughts-on-the-potential-to-compensate-a","title":"Thoughts on the potential to compensate a hearing loss in noise","date":"2021-02-24","arxiv_id":"2102.12397","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-optimization-of-contexts-for","title":"Evolutionary optimization of contexts for phonetic correction in speech recognition systems","date":"2021-02-23","arxiv_id":"2102.11480","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-human-readable-transcript-for","title":"Generating Human Readable Transcript for Automatic Speech Recognition with Pre-trained Language Model","date":"2021-02-22","arxiv_id":"2102.11114","repositories_listed":0,"syntology":null},{"url":null,"slug":"echo-state-speech-recognition","title":"Echo State Speech Recognition","date":"2021-02-18","arxiv_id":"2102.09114","repositories_listed":0,"syntology":null},{"url":null,"slug":"fundamental-frequency-feature-normalization","title":"Fundamental Frequency Feature Normalization and Data Augmentation for Child Speech Recognition","date":"2021-02-18","arxiv_id":"2102.09106","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaussian-kernelized-self-attention-for-long","title":"Gaussian Kernelized Self-Attention for Long Sequence Data and Its Application to CTC-based Speech Recognition","date":"2021-02-18","arxiv_id":"2102.09168","repositories_listed":0,"syntology":null},{"url":null,"slug":"atcspeechnet-a-multilingual-end-to-end-speech","title":"ATCSpeechNet: A multilingual end-to-end speech recognition framework for air traffic control systems","date":"2021-02-17","arxiv_id":"2102.08535","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-multi-source-localization","title":"Deep Learning based Multi-Source Localization with Source Splitting and its Effectiveness in Multi-Talker Speech Recognition","date":"2021-02-16","arxiv_id":"2102.07955","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-automatic-speech-recognition-with","title":"End-to-End Automatic Speech Recognition with Deep Mutual Learning","date":"2021-02-16","arxiv_id":"2102.08154","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-transformer-based-large-context","title":"Hierarchical Transformer-based Large-Context End-to-end ASR with Large-Context Knowledge Distillation","date":"2021-02-16","arxiv_id":"2102.07935","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-models-with","title":"Improving speech recognition models with small samples for air traffic control systems","date":"2021-02-16","arxiv_id":"2102.08015","repositories_listed":0,"syntology":null},{"url":null,"slug":"thank-you-for-attention-a-survey-on-attention","title":"Thank you for Attention: A survey on Attention-based Artificial Neural Networks for Automatic Speech Recognition","date":"2021-02-14","arxiv_id":"2102.07259","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-apc-bidirectional-autoregressive","title":"Bi-APC: Bidirectional Autoregressive Predictive Coding for Unsupervised Pre-training and Its Application to Children's ASR","date":"2021-02-12","arxiv_id":"2102.06816","repositories_listed":0,"syntology":null},{"url":null,"slug":"content-aware-speaker-embeddings-for-speaker","title":"Content-Aware Speaker Embeddings for Speaker Diarisation","date":"2021-02-12","arxiv_id":"2102.06467","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-as-i-mean-not-as-i-say-sequence-loss","title":"Do as I mean, not as I say: Sequence Loss Training for Spoken Language Understanding","date":"2021-02-12","arxiv_id":"2102.06750","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-punctuation-prediction-with","title":"Multimodal Punctuation Prediction with Contextual Dropout","date":"2021-02-12","arxiv_id":"2102.11012","repositories_listed":0,"syntology":null},{"url":null,"slug":"nuva-a-naming-utterance-verifier-for-aphasia","title":"NUVA: A Naming Utterance Verifier for Aphasia Treatment","date":"2021-02-10","arxiv_id":"2102.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsification-via-compressed-sensing-for","title":"Sparsification via Compressed Sensing for Automatic Speech Recognition","date":"2021-02-09","arxiv_id":"2102.04932","repositories_listed":0,"syntology":null},{"url":null,"slug":"train-your-classifier-first-cascade-neural","title":"Train your classifier first: Cascade Neural Networks Training from upper layers to lower layers","date":"2021-02-09","arxiv_id":"2102.04697","repositories_listed":0,"syntology":null}],"record_sha256":"14c4ce69bb55ccb2ca90bff8b9ce54477a9f7a1d9381c64dd7e0603f8b6e3b87","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}