{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/44","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":44,"pages_in_order":65,"rows_per_page":100,"rows":[4301,4400],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/43","next":"/task/speech-recognition/papers/45","papers":[{"url":null,"slug":"massively-multilingual-asr-50-languages-1","title":"Massively Multilingual ASR: 50 Languages, 1 Model, 1 Billion Parameters","date":"2020-07-06","arxiv_id":"2007.03001","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-graph-random-process-for-relational","title":"Deep Graph Random Process for Relational-Thinking-Based Speech Recognition","date":"2020-07-04","arxiv_id":"2007.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-prediction-of-punctuation-and-1","title":"Robust Prediction of Punctuation and Truecasing for Medical ASR","date":"2020-07-04","arxiv_id":"2007.02025","repositories_listed":0,"syntology":null},{"url":null,"slug":"cachenet-a-model-caching-framework-for-deep","title":"CacheNet: A Model Caching Framework for Deep Learning Inference on the Edge","date":"2020-07-03","arxiv_id":"2007.01793","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretrained-semantic-speech-embeddings-for-end","title":"Pretrained Semantic Speech Embeddings for End-to-End Spoken Language Understanding via Cross-Modal Teacher-Student Learning","date":"2020-07-03","arxiv_id":"2007.01836","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-natural-language-processing-1","title":"Applications of Natural Language Processing in Bilingual Language Teaching: An Indonesian-English Case Study","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cuni-neural-asr-with-phoneme-level","title":"CUNI Neural ASR with Phoneme-Level Intermediate Step for\\textasciitildeNon-Native\\textasciitildeSLT at IWSLT 2020","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/end-to-end-offline-speech-translation-system","slug":"end-to-end-offline-speech-translation-system","title":"End-to-End Offline Speech Translation System for IWSLT 2020 using Modality Agnostic Meta-Learning","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-accents-confound-probing-for-accent","title":"How Accents Confound: Probing for Accent Information in End-to-End Speech Recognition Systems","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-effect-of-auxiliary","title":"Investigating the effect of auxiliary objectives for the automated grading of learner English speech transcriptions","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"large-vocabulary-read-speech-corpora-for-four-1","title":"Large Vocabulary Read Speech Corpora for Four Ethiopian Languages: Amharic, Tigrigna, Oromo, and Wolaytta","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lstm-and-gpt-2-synthetic-speech-transfer","title":"LSTM and GPT-2 Synthetic Speech Transfer Learning for Speaker Recognition to Overcome Data Scarcity","date":"2020-07-01","arxiv_id":"2007.00659","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-variational-information-bottleneck","title":"Multi-Task Variational Information Bottleneck","date":"2020-07-01","arxiv_id":"2007.00339","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-and-multiresolution-speech","title":"Multimodal and Multiresolution Speech Recognition with Transformers","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-neural-machine-translation-with-asr","title":"Robust Neural Machine Translation with ASR Errors","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simulspeech-end-to-end-simultaneous-speech-to","title":"SimulSpeech: End-to-End Simultaneous Speech to Text Translation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"start-before-end-and-end-to-end-neural-speech","title":"Start-Before-End and End-to-End: Neural Speech Translation by AppTek and RWTH Aachen University","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-afrl-iwslt-2020-systems-work-from-home","title":"The AFRL IWSLT 2020 Systems: Work-From-Home Edition","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tigrinya-automatic-speech-recognition-with","title":"Tigrinya Automatic Speech recognition with Morpheme based recognition units","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-end-2-end-learning-for-predicting","title":"Towards end-2-end learning for predicting behavior codes from spoken utterances in psychotherapy conversations","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-stream-translation-adaptive","title":"Towards Stream Translation: Adaptive Computation Time for Simultaneous Machine Translation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-asr-error-correction","title":"Towards Understanding ASR Error Correction for Medical Conversations","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-frequency-lstm-an-efficient","title":"Multi-view Frequency LSTM: An Efficient Frontend for Automatic Speech Recognition","date":"2020-06-30","arxiv_id":"2007.00131","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-machine-translation-for-multilingual","title":"Neural Machine Translation for Multilingual Grapheme-to-Phoneme Conversion","date":"2020-06-25","arxiv_id":"2006.14194","repositories_listed":0,"syntology":null},{"url":"/paper/sequence-to-multi-sequence-learning-via","slug":"sequence-to-multi-sequence-learning-via","title":"Sequence to Multi-Sequence Learning via Conditional Chain Mapping for Mixture Signals","date":"2020-06-25","arxiv_id":"2006.14150","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-transformer-asr-with-blockwise","title":"Streaming Transformer ASR with Blockwise Synchronous Inference","date":"2020-06-25","arxiv_id":"2006.14941","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-model-to-pronounce-them-all-multilingual-1","title":"One Model to Pronounce Them All: Multilingual Grapheme-to-Phoneme Conversion With a Transformer Ensemble","date":"2020-06-23","arxiv_id":"2006.13343","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-neural-networks-an-introduction-and","title":"Bayesian Neural Networks: An Introduction and Survey","date":"2020-06-22","arxiv_id":"2006.12024","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-representations-improve-end","title":"Self-Supervised Representations Improve End-to-End Speech Translation","date":"2020-06-22","arxiv_id":"2006.12124","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-double-side-learning-ensemble-model-for","title":"Deep Double-Side Learning Ensemble Model for Few-Shot Parkinson Speech Recognition","date":"2020-06-20","arxiv_id":"2006.11593","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-active-learning-for-automatic","title":"Boosting Active Learning for Speech Recognition with Noisy Pseudo-labeled Samples","date":"2020-06-19","arxiv_id":"2006.11021","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-speaker-counting-speech-recognition-and","title":"Joint Speaker Counting, Speech Recognition, and Speaker Identification for Overlapped Speech of Any Number of Speakers","date":"2020-06-19","arxiv_id":"2006.10930","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-code-switching-language-models-for","title":"End-to-End Code Switching Language Models for Automatic Speech Recognition","date":"2020-06-16","arxiv_id":"2006.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-of-acoustic-model-parameters-in","title":"Quantization of Acoustic Model Parameters in Automatic Speech Recognition Framework","date":"2020-06-16","arxiv_id":"2006.09054","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automated-assessment-of-stuttering","title":"Towards Automated Assessment of Stuttering and Stuttering Therapy","date":"2020-06-16","arxiv_id":"2006.09222","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-end-to-end-asr-for-openstt","title":"Exploration of End-to-End ASR for OpenSTT -- Russian Open Speech-to-Text Dataset","date":"2020-06-15","arxiv_id":"2006.08274","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-forward-backward-decoder-for","title":"Regularized Forward-Backward Decoder for Attention Models","date":"2020-06-15","arxiv_id":"2006.08506","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-jhu-multi-microphone-multi-speaker-asr","title":"The JHU Multi-Microphone Multi-Speaker ASR System for the CHiME-6 Challenge","date":"2020-06-14","arxiv_id":"2006.07898","repositories_listed":0,"syntology":null},{"url":null,"slug":"uwspeech-speech-to-speech-translation-for","title":"UWSpeech: Speech to Speech Translation for Unwritten Languages","date":"2020-06-14","arxiv_id":"2006.07926","repositories_listed":0,"syntology":null},{"url":null,"slug":"frugalml-how-to-use-ml-prediction-apis-more","title":"FrugalML: How to Use ML Prediction APIs More Accurately and Cheaply","date":"2020-06-12","arxiv_id":"2006.07512","repositories_listed":0,"syntology":null},{"url":null,"slug":"notic-my-speech-blending-speech-patterns-with","title":"\"Notic My Speech\" -- Blending Speech Patterns With Multimedia","date":"2020-06-12","arxiv_id":"2006.08599","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-training-dialog-models","title":"Data Augmentation for Training Dialog Models Robust to Speech Recognition Errors","date":"2020-06-10","arxiv_id":"2006.05635","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-lingual-transfer-learning-for","title":"Improving Cross-Lingual Transfer Learning for End-to-End Speech Recognition with Speech Translation","date":"2020-06-09","arxiv_id":"2006.05474","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-not-to-discriminate-task-agnostic","title":"Learning not to Discriminate: Task Agnostic Learning for Improving Monolingual and Code-switched Speech Recognition","date":"2020-06-09","arxiv_id":"2006.05257","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-neural-text","title":"On the Effectiveness of Neural Text Generation based Data Augmentation for Recognition of Morphologically Rich Speech","date":"2020-06-09","arxiv_id":"2006.05129","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-the-weight-space-for-different","title":"Characterizing the Weight Space for Different Learning Models","date":"2020-06-04","arxiv_id":"2006.02724","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-rnn-t-for-open-domain-asr","title":"Contextual RNN-T For Open Domain ASR","date":"2020-06-04","arxiv_id":"2006.03411","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-talker-asr-for-an-unknown-number-of","title":"Multi-talker ASR for an unknown number of sources: Joint training of source counting, separation and ASR","date":"2020-06-04","arxiv_id":"2006.02786","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-training-for-end-to-end-speech-1","title":"Self-Training for End-to-End Speech Translation","date":"2020-06-03","arxiv_id":"2006.02490","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-for-british-sign-language-1","title":"Transfer Learning for British Sign Language Modelling","date":"2020-06-03","arxiv_id":"2006.02144","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-quality-and-stability-of-a","title":"Analyzing the Quality and Stability of a Streaming End-to-End On-Device Speech Recognizer","date":"2020-06-02","arxiv_id":"2006.01416","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-audio-attacks-on-asr-systems-with","title":"Detecting Audio Attacks on ASR Systems with Dropout Uncertainty","date":"2020-06-02","arxiv_id":"2006.01906","repositories_listed":0,"syntology":null},{"url":null,"slug":"dilated-u-net-based-approach-for-multichannel","title":"Dilated U-net based approach for multichannel speech enhancement from First-Order Ambisonics recordings","date":"2020-06-02","arxiv_id":"2006.01708","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-contextual-language-modeling","title":"An Effective Contextual Language Modeling Framework for Speech Summarization with Augmented Features","date":"2020-06-01","arxiv_id":"2006.01189","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyse-de-l-effet-de-la-r-everb-eration-sur","title":"Analyse de l'effet de la r\\'everb\\'eration sur la reconnaissance automatique de la parole (Analyzing how reverberation affects Automatic Speech Recognition)","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-variational-autoencoder-for","title":"Constrained Variational Autoencoder for improving EEG based Speech Recognition Systems","date":"2020-06-01","arxiv_id":"2006.02902","repositories_listed":0,"syntology":null},{"url":null,"slug":"introduction-d-informations-s-emantiques-dans","title":"Introduction d'informations s\\'emantiques dans un syst\\`eme de reconnaissance de la parole (Despite spectacular advances in recent years, the Automatic Speech Recognition (ASR) systems still make mistakes, especially in noisy environments)","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-recognize-code-switched-speech","title":"Learning to Recognize Code-switched Speech Without Forgetting Monolingual Speech Recognition","date":"2020-06-01","arxiv_id":"2006.00782","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconnaissance-automatique-de-la-parole-g-en","title":"Reconnaissance automatique de la parole : g\\'en\\'eration des prononciations non natives pour l'enrichissement du lexique (In this study we propose a method for lexicon adaptation in order to improve the automatic speech recognition (ASR) of non-native speakers)","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-language-identification-using","title":"Streaming Language Identification using Combination of Acoustic Representations and ASR Hypotheses","date":"2020-06-01","arxiv_id":"2006.00703","repositories_listed":0,"syntology":null},{"url":null,"slug":"sur-l-utilisation-de-la-reconnaissance","title":"Sur l'utilisation de la reconnaissance automatique de la parole pour l'aide au diagnostic diff\\'erentiel entre la maladie de Parkinson et l'AMS (On using automatic speech recognition for the differential diagnosis of Parkinson's Disease and MSA This article presents a study regarding the contribution of automatic speech processing in the differential diagnosis between Parkinson's disease and MSA (Multi-System Atrophies))","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-masking-for-improved-stability-in","title":"Dynamic Masking for Improved Stability in Spoken Language Translation","date":"2020-05-30","arxiv_id":"2006.00249","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-eeg-based-continuous-speech-1","title":"Improving EEG based continuous speech recognition using GAN","date":"2020-05-29","arxiv_id":"2006.01260","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-effect-of-speech-perception-in","title":"Understanding effect of speech perception in EEG based speech recognition systems","date":"2020-05-29","arxiv_id":"2006.01261","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-and-defense-on-textual","title":"Adversarial Attacks and Defense on Texts: A Survey","date":"2020-05-28","arxiv_id":"2005.14108","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-can-self-attention-be-replaced-by-feed","title":"When Can Self-Attention Be Replaced by Feed Forward Layers?","date":"2020-05-28","arxiv_id":"2005.13895","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-entity-popularity-to-improve","title":"Predicting Entity Popularity to Improve Spoken Entity Recognition by Virtual Assistants","date":"2020-05-26","arxiv_id":"2005.12816","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-audio-enriched-bert-based-framework-for","title":"An Audio-enriched BERT-based Framework for Spoken Multiple-choice Question Answering","date":"2020-05-25","arxiv_id":"2005.12142","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-mispronunciation-detection","title":"An End-to-End Mispronunciation Detection System for L2 English Speech Leveraging Novel Anti-Phone Modeling","date":"2020-05-25","arxiv_id":"2005.11950","repositories_listed":0,"syntology":null},{"url":"/paper/ft-speech-danish-parliament-speech-corpus","slug":"ft-speech-danish-parliament-speech-corpus","title":"FT Speech: Danish Parliament Speech Corpus","date":"2020-05-25","arxiv_id":"2005.12368","repositories_listed":0,"syntology":null},{"url":"/paper/asapp-asr-multistream-cnn-and-self-attentive","slug":"asapp-asr-multistream-cnn-and-self-attentive","title":"ASAPP-ASR: Multistream CNN and Self-Attentive SRU for SOTA Speech Recognition","date":"2020-05-21","arxiv_id":"2005.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-evaluation-of-importance-maps-in","title":"Large scale evaluation of importance maps in automatic speech recognition","date":"2020-05-21","arxiv_id":"2005.10929","repositories_listed":0,"syntology":null},{"url":null,"slug":"multistream-cnn-for-robust-acoustic-modeling","title":"Multistream CNN for Robust Acoustic Modeling","date":"2020-05-21","arxiv_id":"2005.10470","repositories_listed":0,"syntology":null},{"url":null,"slug":"simplified-self-attention-for-transformer","title":"Simplified Self-Attention for Transformer-based End-to-End Speech Recognition","date":"2020-05-21","arxiv_id":"2005.10463","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-label-synchronous-and-frame","title":"A Comparison of Label-Synchronous and Frame-Synchronous End-to-End Models for Speech Recognition","date":"2020-05-20","arxiv_id":"2005.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"early-stage-lm-integration-using-local-and","title":"Early Stage LM Integration Using Local and Global Log-Linear Combination","date":"2020-05-20","arxiv_id":"2005.10049","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-large-margin-softmax-in","title":"Investigation of Large-Margin Softmax in Neural Language Modeling","date":"2020-05-20","arxiv_id":"2005.10089","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-positional-encoding-for-speech","title":"Relative Positional Encoding for Speech Recognition and Direct Translation","date":"2020-05-20","arxiv_id":"2005.09940","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-approaches-for-neural-decoding","title":"Deep learning approaches for neural decoding: from CNNs to LSTMs and spikes to fMRI","date":"2020-05-19","arxiv_id":"2005.09687","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-transformers-for-large-scale-speech","title":"Exploring Transformers for Large-Scale Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09684","repositories_listed":0,"syntology":null},{"url":"/paper/fast-simpler-and-more-accurate-hybrid-asr","slug":"fast-simpler-and-more-accurate-hybrid-asr","title":"Faster, Simpler and More Accurate Hybrid ASR Systems Using Wordpieces","date":"2020-05-19","arxiv_id":"2005.09150","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-proper-noun-recognition-in-end-to","title":"Improving Proper Noun Recognition in End-to-End ASR By Customization of the MWER Loss Criterion","date":"2020-05-19","arxiv_id":"2005.09756","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-end-to-end-modeling-approach-for","title":"An Effective End-to-End Modeling Approach for Mispronunciation Detection","date":"2020-05-18","arxiv_id":"2005.08440","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-transducer-for-online-speech","title":"Attention-based Transducer for Online Speech Recognition","date":"2020-05-18","arxiv_id":"2005.08497","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-multi-channel-recognition-of","title":"Audio-visual Multi-channel Recognition of Overlapped Speech","date":"2020-05-18","arxiv_id":"2005.08571","repositories_listed":0,"syntology":null},{"url":null,"slug":"weak-attention-suppression-for-transformer","title":"Weak-Attention Suppression For Transformer Based Speech Recognition","date":"2020-05-18","arxiv_id":"2005.09137","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-text-adaptation-towards-an","title":"Speech to Text Adaptation: Towards an Efficient Cross-Modal Distillation","date":"2020-05-17","arxiv_id":"2005.08213","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-learning-based-wearable-healthcare-iot","title":"A Deep Learning based Wearable Healthcare IoT Device for AI-enabled Hearing Assistance Automation","date":"2020-05-16","arxiv_id":"2005.08076","repositories_listed":0,"syntology":null},{"url":"/paper/accentdb-a-database-of-non-native-english","slug":"accentdb-a-database-of-non-native-english","title":"AccentDB: A Database of Non-Native English Accents to Assist Neural Speech Recognition","date":"2020-05-16","arxiv_id":"2005.07973","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-sparsity-neural-networks-for","title":"Dynamic Sparsity Neural Networks for Automatic Speech Recognition","date":"2020-05-16","arxiv_id":"2005.10627","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-weakly-and-semi-supervised","title":"Large scale weakly and semi-supervised learning for low-resource video ASR","date":"2020-05-16","arxiv_id":"2005.07850","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-spelling-inconsistencies-in-code","title":"Reducing Spelling Inconsistencies in Code-Switching ASR using Contextualized CTC Loss","date":"2020-05-16","arxiv_id":"2005.07920","repositories_listed":0,"syntology":null},{"url":null,"slug":"spike-triggered-non-autoregressive","title":"Spike-Triggered Non-Autoregressive Transformer for End-to-End Speech Recognition","date":"2020-05-16","arxiv_id":"2005.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"that-sounds-familiar-an-analysis-of-phonetic","title":"That Sounds Familiar: an Analysis of Phonetic Representations Transfer Across Languages","date":"2020-05-16","arxiv_id":"2005.08118","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-dependent-acoustic-modeling-without","title":"Context-Dependent Acoustic Modeling without Explicit Phone Clustering","date":"2020-05-15","arxiv_id":"2005.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualizing-asr-lattice-rescoring-with","title":"Contextualizing ASR Lattice Rescoring with Hybrid Pointer Network Language Model","date":"2020-05-15","arxiv_id":"2005.07394","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-do-not-need-more-data-improving-end-to","title":"You Do Not Need More Data: Improving End-To-End Speech Recognition by Text-To-Speech Data Augmentation","date":"2020-05-14","arxiv_id":"2005.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"darts-asr-differentiable-architecture-search","title":"DARTS-ASR: Differentiable Architecture Search for Multilingual Speech Recognition and Adaptation","date":"2020-05-13","arxiv_id":"2005.07029","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-estimation-of-inteligibility","title":"Automatic Estimation of Intelligibility Measure for Consonants in Speech","date":"2020-05-12","arxiv_id":"2005.06065","repositories_listed":0,"syntology":null},{"url":null,"slug":"discretalk-text-to-speech-as-a-machine","title":"DiscreTalk: Text-to-Speech as a Machine Translation Problem","date":"2020-05-12","arxiv_id":"2005.05525","repositories_listed":0,"syntology":null}],"record_sha256":"b144d26b92530cc74263630bff64c4018a51ab7c29b004ee8a37fc61e21f5fbf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}