{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/22","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":22,"pages_in_order":31,"rows_per_page":100,"rows":[2101,2200],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/21","next":"/task/automatic-speech-recognition/papers/23","papers":[{"url":null,"slug":"a-bandit-approach-to-curriculum-generation","title":"A bandit approach to curriculum generation for automatic speech recognition","date":"2021-02-06","arxiv_id":"2102.03662","repositories_listed":0,"syntology":null},{"url":null,"slug":"intermediate-loss-regularization-for-ctc","title":"Intermediate Loss Regularization for CTC-based Speech Recognition","date":"2021-02-05","arxiv_id":"2102.03216","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-self-supervised-pre-training-for","title":"Multi-Task Self-Supervised Pre-Training for Music Classification","date":"2021-02-05","arxiv_id":"2102.03229","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-augmentation-and-adaptive-ctc","title":"Two-Stage Augmentation and Adaptive CTC Fusion for Improved Robustness of Multi-Stream End-to-End ASR","date":"2021-02-05","arxiv_id":"2102.03055","repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-number-of-filters-of-convolutional","title":"Effects of Number of Filters of Convolutional Layers on Speech Recognition Model Accuracy","date":"2021-02-03","arxiv_id":"2102.02326","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-language-model-training-for-domain","title":"Internal Language Model Training for Domain-Adaptive End-to-End Speech Recognition","date":"2021-02-02","arxiv_id":"2102.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-by-simply-fine-tuning-bert","title":"Speech Recognition by Simply Fine-tuning BERT","date":"2021-01-30","arxiv_id":"2102.00291","repositories_listed":0,"syntology":null},{"url":null,"slug":"bcn2brno-asr-system-fusion-for-albayzin-2020","title":"BCN2BRNO: ASR System Fusion for Albayzin 2020 Speech to Text Challenge","date":"2021-01-29","arxiv_id":"2101.12729","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-end-to-end-asr-for-endangered","title":"Leveraging End-to-End ASR for Endangered Language Documentation: An Empirical Study on Yoloxóchitl Mixtec","date":"2021-01-26","arxiv_id":"2101.10877","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-beam-search-confidence-for-energy","title":"Exploiting Beam Search Confidence for Energy-Efficient Speech Recognition","date":"2021-01-22","arxiv_id":"2101.09083","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-models-for-joint-speech-recognition","title":"Streaming Models for Joint Speech Recognition and Translation","date":"2021-01-22","arxiv_id":"2101.09149","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-wav2vec2-0-and-bert-into-end-to-end","title":"Efficiently Fusing Pretrained Acoustic and Linguistic Encoders for Low-resource Speech Recognition","date":"2021-01-17","arxiv_id":"2101.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evaluation-of-word-level-confidence","title":"An evaluation of word-level confidence estimation for end-to-end automatic speech recognition","date":"2021-01-14","arxiv_id":"2101.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-offline-transformer-based-end-to-end","title":"Fast offline Transformer-based end-to-end automatic speech recognition for real-world applications","date":"2021-01-14","arxiv_id":"2101.05600","repositories_listed":0,"syntology":null},{"url":null,"slug":"wer-bert-automatic-wer-estimation-with-bert","title":"WER-BERT: Automatic WER Estimation with BERT in a Balanced Ordinal Classification Paradigm","date":"2021-01-14","arxiv_id":"2101.05478","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypothesis-stitcher-for-end-to-end-speaker","title":"Hypothesis Stitcher for End-to-End Speaker-attributed ASR on Long-form Multi-talker Recordings","date":"2021-01-06","arxiv_id":"2101.01853","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-without-forgetting-task-aware","title":"Learning without Forgetting: Task Aware Multitask Learning for Multi-Modality Tasks","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nas-bench-asr-reproducible-neural","title":"NAS-Bench-ASR: Reproducible Neural Architecture Search for Speech Recognition","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"why-does-decentralized-training-outperform","title":"Why Does Decentralized Training Outperform Synchronous Training In The Large Batch Setting?","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-multi-frame-adl-mvdr-for-target","title":"Multi-channel Multi-frame ADL-MVDR for Target Speech Separation","date":"2020-12-24","arxiv_id":"2012.13442","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-reasoning-graph-neural-network","title":"A Hierarchical Reasoning Graph Neural Network for The Automatic Scoring of Answer Transcriptions in Video Job Interviews","date":"2020-12-22","arxiv_id":"2012.11960","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-meta-sampling-for-multilingual","title":"Adversarial Meta Sampling for Multilingual Low-Resource Speech Recognition","date":"2020-12-22","arxiv_id":"2012.11896","repositories_listed":0,"syntology":null},{"url":null,"slug":"adjust-free-adversarial-example-generation-in","title":"Adjust-free adversarial example generation in speech recognition using evolutionary multi-objective optimization under black-box condition","date":"2020-12-21","arxiv_id":"2012.11138","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-streaming-asr-with-non-autoregressive","title":"Toward Streaming ASR with Non-Autoregressive Insertion-based Model","date":"2020-12-18","arxiv_id":"2012.10128","repositories_listed":0,"syntology":null},{"url":"/paper/exploring-transfer-learning-for-end-to-end","slug":"exploring-transfer-learning-for-end-to-end","title":"Exploring Transfer Learning For End-to-End Spoken Language Understanding","date":"2020-12-15","arxiv_id":"2012.08549","repositories_listed":0,"syntology":null},{"url":null,"slug":"user-friendly-automatic-transcription-of-low","title":"User-friendly automatic transcription of low-resource languages: Plugging ESPnet into Elpis","date":"2020-12-15","arxiv_id":"2101.03027","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-on-device-fully-neural-end-to-end","title":"A review of on-device fully neural end-to-end automatic speech recognition algorithms","date":"2020-12-14","arxiv_id":"2012.07974","repositories_listed":0,"syntology":null},{"url":null,"slug":"less-is-more-improved-rnn-t-decoding-using","title":"Less Is More: Improved RNN-T Decoding Using Limited Label Context and Path Merging","date":"2020-12-12","arxiv_id":"2012.06749","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-robustness-to-disfluencies-in-rnn","title":"Improved Robustness to Disfluencies in RNN-Transducer Based Speech Recognition","date":"2020-12-11","arxiv_id":"2012.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-learning-of-lf-mmi-trained-time","title":"Bayesian Learning of LF-MMI Trained Time Delay Neural Networks for Speech Recognition","date":"2020-12-08","arxiv_id":"2012.04494","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-multiple-asr-hypotheses-to-boost-i18n","title":"Using multiple ASR hypotheses to boost i18n NLU performance","date":"2020-12-07","arxiv_id":"2012.04099","repositories_listed":0,"syntology":null},{"url":"/paper/100000-podcasts-a-spoken-english-document","slug":"100000-podcasts-a-spoken-english-document","title":"100,000 Podcasts: A Spoken English Document Corpus","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-for-non-standardised-languages-with","title":"ASR for Non-standardised Languages with Dialectal Variation: the case of Swiss German","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"german-arabic-speech-to-speech-translation","title":"German-Arabic Speech-to-Speech Translation for Psychiatric Diagnosis","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-of-spoken-language","title":"Multi-task Learning of Spoken Language Understanding by Integrating N-Best Hypotheses with Hierarchical Attention","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-detection-of-sentence-completion","title":"On-Device detection of sentence completion for voice assistants with low-memory footprint","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-transcription","title":"Sparse Transcription","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-indigenous-languages-technology-project","title":"The Indigenous Languages Technology project at NRC Canada: An empowerment-oriented approach to developing language software","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accuracy-of-rare-words-for-rnn","title":"Improving accuracy of rare words for RNN-Transducer through unigram shallow fusion","date":"2020-11-30","arxiv_id":"2012.00133","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-transducers-for-code-switched","title":"Transformer-Transducers for Code-Switched Speech Recognition","date":"2020-11-30","arxiv_id":"2011.15023","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-for-speech","title":"Unsupervised Domain Adaptation for Speech Recognition via Uncertainty Driven Self-Training","date":"2020-11-26","arxiv_id":"2011.13439","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrap-an-end-to-end-asr-system-by","title":"Bootstrap an end-to-end ASR system by multilingual training, transfer learning, text-to-text mapping and synthetic audio","date":"2020-11-25","arxiv_id":"2011.12696","repositories_listed":0,"syntology":null},{"url":null,"slug":"adam-a-stochastic-method-with-adaptive-1","title":"Adam$^+$: A Stochastic Method with Adaptive Variance Reduction","date":"2020-11-24","arxiv_id":"2011.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-language-modeling-for-improving","title":"Multi-task Language Modeling for Improving Speech Recognition of Rare Words","date":"2020-11-23","arxiv_id":"2011.11715","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-synthetic-audio-to-improve-the","title":"Using Synthetic Audio to Improve The Recognition of Out-Of-Vocabulary Words in End-To-End ASR Systems","date":"2020-11-23","arxiv_id":"2011.11564","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-t-asr-accuracy-using-context","title":"Improving RNN-T ASR Accuracy Using Context Audio","date":"2020-11-20","arxiv_id":"2011.10538","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascade-rnn-transducer-syllable-based","title":"Cascade RNN-Transducer: Syllable Based Streaming On-device Mandarin Speech Recognition with a Syllable-to-Character Converter","date":"2020-11-17","arxiv_id":"2011.08469","repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-automatic-speech-recognition-system","title":"Refining Automatic Speech Recognition System for older adults","date":"2020-11-17","arxiv_id":"2011.08346","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-multi-channel-integration-and","title":"Audio-visual Multi-channel Integration and Recognition of Overlapped Speech","date":"2020-11-16","arxiv_id":"2011.07755","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-shallow-fusion-for-rnn-t-personalization","title":"Deep Shallow Fusion for RNN-T Personalization","date":"2020-11-16","arxiv_id":"2011.07754","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-guided-by-contextual","title":"Improving Speech Enhancement Performance by Leveraging Contextual Broad Phonetic Class Information","date":"2020-11-15","arxiv_id":"2011.07442","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-reinforcement-learning-for","title":"Self-supervised reinforcement learning for speaker localisation with the iCub humanoid robot","date":"2020-11-12","arxiv_id":"2011.06544","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-speech-to-speech-translation","title":"Simultaneous Speech-to-Speech Translation System with Neural Incremental ASR, MT, and TTS","date":"2020-11-10","arxiv_id":"2011.04845","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-lf-mmi-ctc-and-rnn-t-criteria","title":"Benchmarking LF-MMI, CTC and RNN-T Criteria for Streaming ASR","date":"2020-11-09","arxiv_id":"2011.04785","repositories_listed":0,"syntology":null},{"url":null,"slug":"gated-recurrent-fusion-with-joint-training","title":"Gated Recurrent Fusion with Joint Training Framework for Robust End-to-End Speech Recognition","date":"2020-11-09","arxiv_id":"2011.04249","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-query-rewriting-in","title":"Personalized Query Rewriting in Conversational AI Agents","date":"2020-11-09","arxiv_id":"2011.04748","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-usefulness-of-self-attention-for","title":"On the Usefulness of Self-Attention for Automatic Speech Recognition with Transformers","date":"2020-11-08","arxiv_id":"2011.04906","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustics-based-intent-recognition-using","title":"Acoustics Based Intent Recognition Using Discovered Phonetic Units for Low Resource Languages","date":"2020-11-07","arxiv_id":"2011.03646","repositories_listed":0,"syntology":null},{"url":null,"slug":"espnet-se-end-to-end-speech-enhancement-and","title":"ESPnet-se: end-to-end speech enhancement and separation toolkit designed for asr integration","date":"2020-11-07","arxiv_id":"2011.03706","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-restricted-streaming-recurrent","title":"Alignment Restricted Streaming Recurrent Neural Network Transducer","date":"2020-11-05","arxiv_id":"2011.03072","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-images-for-asr-and-tts-through","title":"Augmenting Images for ASR and TTS through Single-loop and Dual-loop Multimodal Chain Framework","date":"2020-11-04","arxiv_id":"2011.02099","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-end-to-end-code","title":"Data Augmentation for End-to-end Code-switching Speech Recognition","date":"2020-11-04","arxiv_id":"2011.02160","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-machine-speech-chain-towards","title":"Incremental Machine Speech Chain Towards Enabling Listening while Speaking in Real-time","date":"2020-11-04","arxiv_id":"2011.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-to-sequence-learning-via-attention","title":"Sequence-to-Sequence Learning via Attention Transfer for Incremental Speech Recognition","date":"2020-11-04","arxiv_id":"2011.02127","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-transducer-with-normalized","title":"Improving RNN transducer with normalized jointer network","date":"2020-11-03","arxiv_id":"2011.01576","repositories_listed":0,"syntology":null},{"url":null,"slug":"integration-of-speech-separation-diarization","title":"Integration of speech separation, diarization, and recognition for multi-speaker meetings: System description, comparison, and analysis","date":"2020-11-03","arxiv_id":"2011.02014","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-language-model-estimation-for-domain","title":"Internal Language Model Estimation for Domain-Adaptive End-to-End Speech Recognition","date":"2020-11-03","arxiv_id":"2011.01991","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-attention-based-models-with","title":"Streaming Attention-Based Models with Augmented Memory for End-to-End Speech Recognition","date":"2020-11-03","arxiv_id":"2011.07120","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-pattern-discovery-from-thematic","title":"Unsupervised Pattern Discovery from Thematic Speech Archives Based on Multilingual Bottleneck Features","date":"2020-11-03","arxiv_id":"2011.01986","repositories_listed":0,"syntology":null},{"url":null,"slug":"warped-language-models-for-noise-robust","title":"Warped Language Models for Noise Robust Language Understanding","date":"2020-11-03","arxiv_id":"2011.01900","repositories_listed":0,"syntology":null},{"url":null,"slug":"dnn-based-semantic-model-for-rescoring-n-best","title":"DNN-Based Semantic Model for Rescoring N-best Speech Recognition List","date":"2020-11-02","arxiv_id":"2011.00975","repositories_listed":0,"syntology":null},{"url":null,"slug":"sapaugment-learning-a-sample-adaptive-policy","title":"SapAugment: Learning A Sample Adaptive Policy for Data Augmentation","date":"2020-11-02","arxiv_id":"2011.01156","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectively-pretraining-a-speech-translation","title":"Effectively pretraining a speech translation decoder with Machine Translation data","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"elitr-european-live-translator","title":"ELITR: European Live Translator","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-asr-on-alzheimers-disease-detection-1","title":"Impact of ASR on Alzheimer’s Disease Detection: All Errors are Equal, but Deletions are More Equal than Others","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-end-to-end-bangla-speech","title":"Improving End-to-End Bangla Speech Recognition with Semi-supervised Training","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"directional-asr-a-new-paradigm-for-e2e-multi","title":"Directional ASR: A New Paradigm for E2E Multi-Speaker Speech Recognition with Source Localization","date":"2020-10-30","arxiv_id":"2011.00091","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-simultaneous-speech-translation","title":"Streaming Simultaneous Speech Translation with Augmented Memory Transformer","date":"2020-10-30","arxiv_id":"2011.00033","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-speech-recognition-via-graph","title":"Semi-Supervised Speech Recognition via Graph-based Temporal Classification","date":"2020-10-29","arxiv_id":"2010.15653","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupling-pronunciation-and-language-for-end","title":"Decoupling Pronunciation and Language for End-to-end Code-switching Automatic Speech Recognition","date":"2020-10-28","arxiv_id":"2010.14798","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusion-models-for-improved-visual-captioning","title":"Fusion Models for Improved Visual Captioning","date":"2020-10-28","arxiv_id":"2010.15251","repositories_listed":0,"syntology":null},{"url":null,"slug":"int8-winograd-acceleration-for-conv1d","title":"INT8 Winograd Acceleration for Conv1D Equipped ASR Models Deployed on Mobile Devices","date":"2020-10-28","arxiv_id":"2010.14841","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-transformer-asr-with-ctc","title":"Non-Autoregressive Transformer ASR with CTC-Enhanced Decoder Input","date":"2020-10-28","arxiv_id":"2010.15025","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-encoders-for-unifying-streaming-and","title":"Cascaded encoders for unifying streaming and non-streaming ASR","date":"2020-10-27","arxiv_id":"2010.14606","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-decoder-masking-for-transformer","title":"Effective Decoder Masking for Transformer Based End-to-End Speech Recognition","date":"2020-10-27","arxiv_id":"2010.14764","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotion-recognition-by-fusing-time","title":"Emotion recognition by fusing time synchronous and time asynchronous representations","date":"2020-10-27","arxiv_id":"2010.14102","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-mask-ctc-for-non-autoregressive-end","title":"Improved Mask-CTC for Non-Autoregressive End-to-End ASR","date":"2020-10-26","arxiv_id":"2010.13270","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-noise-robustness-of-an-end-to-end","title":"Improving Noise Robustness of an End-to-End Neural Model for Automatic Speech Recognition","date":"2020-10-23","arxiv_id":"2010.12715","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-streaming-automatic-speech","title":"Improving Streaming Automatic Speech Recognition With Non-Streaming Model Distillation On Unsupervised Data","date":"2020-10-22","arxiv_id":"2010.12096","repositories_listed":0,"syntology":null},{"url":null,"slug":"mam-masked-acoustic-modeling-for-end-to-end","title":"MAM: Masked Acoustic Modeling for End-to-End Speech-to-Text Translation","date":"2020-10-22","arxiv_id":"2010.11445","repositories_listed":0,"syntology":null},{"url":null,"slug":"slimipl-language-model-free-iterative-pseudo","title":"SlimIPL: Language-Model-Free Iterative Pseudo-Labeling","date":"2020-10-22","arxiv_id":"2010.11524","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-multi-task-learning-framework-to","title":"A General Multi-Task Learning Framework to Leverage Text Data for Speech to Text Tasks","date":"2020-10-21","arxiv_id":"2010.11338","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-models-with-cyclic-feedback-for","title":"Cascaded Models With Cyclic Feedback For Direct Speech Translation","date":"2020-10-21","arxiv_id":"2010.11153","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-improved-accuracy","title":"Knowledge Distillation for Improved Accuracy in Spoken Question Answering","date":"2020-10-21","arxiv_id":"2010.11067","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentence-boundary-augmentation-for-neural","title":"Sentence Boundary Augmentation For Neural Machine Translation Robustness","date":"2020-10-21","arxiv_id":"2010.11132","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-transfer-for-efficient-on-device","title":"Knowledge Transfer for Efficient On-device False Trigger Mitigation","date":"2020-10-20","arxiv_id":"2010.10591","repositories_listed":0,"syntology":null},{"url":null,"slug":"replacing-human-audio-with-synthetic-audio","title":"Replacing Human Audio with Synthetic Audio for On-device Unspoken Punctuation Prediction","date":"2020-10-20","arxiv_id":"2010.10203","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-chinese-end-to-end-spoken-language","title":"Ensemble Chinese End-to-End Spoken Language Understanding for Abnormal Event Detection from audio stream","date":"2020-10-19","arxiv_id":"2010.09235","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-data-distillation-for-end-to-end-1","title":"Towards Data Distillation for End-to-end Spoken Conversational Question Answering","date":"2020-10-18","arxiv_id":"2010.08923","repositories_listed":0,"syntology":null},{"url":null,"slug":"studying-the-similarity-of-covid-19-sounds","title":"Studying the Similarity of COVID-19 Sounds based on Correlation Analysis of MFCC","date":"2020-10-17","arxiv_id":"2010.08770","repositories_listed":0,"syntology":null}],"record_sha256":"5f118b41dcf2135a7a57306827b00a1dda52564b0ec120bef30d33c73e7ec80e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}