{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/25","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":32,"rows_per_page":100,"rows":[2401,2500],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/24","next":"/task/automatic-speech-recognition-2/papers/26","papers":[{"url":null,"slug":"improving-accuracy-of-rare-words-for-rnn","title":"Improving accuracy of rare words for RNN-Transducer through unigram shallow fusion","date":"2020-11-30","arxiv_id":"2012.00133","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-transducers-for-code-switched","title":"Transformer-Transducers for Code-Switched Speech Recognition","date":"2020-11-30","arxiv_id":"2011.15023","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-for-speech","title":"Unsupervised Domain Adaptation for Speech Recognition via Uncertainty Driven Self-Training","date":"2020-11-26","arxiv_id":"2011.13439","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrap-an-end-to-end-asr-system-by","title":"Bootstrap an end-to-end ASR system by multilingual training, transfer learning, text-to-text mapping and synthetic audio","date":"2020-11-25","arxiv_id":"2011.12696","repositories_listed":0,"syntology":null},{"url":null,"slug":"adam-a-stochastic-method-with-adaptive-1","title":"Adam$^+$: A Stochastic Method with Adaptive Variance Reduction","date":"2020-11-24","arxiv_id":"2011.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-language-modeling-for-improving","title":"Multi-task Language Modeling for Improving Speech Recognition of Rare Words","date":"2020-11-23","arxiv_id":"2011.11715","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-synthetic-audio-to-improve-the","title":"Using Synthetic Audio to Improve The Recognition of Out-Of-Vocabulary Words in End-To-End ASR Systems","date":"2020-11-23","arxiv_id":"2011.11564","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-t-asr-accuracy-using-context","title":"Improving RNN-T ASR Accuracy Using Context Audio","date":"2020-11-20","arxiv_id":"2011.10538","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascade-rnn-transducer-syllable-based","title":"Cascade RNN-Transducer: Syllable Based Streaming On-device Mandarin Speech Recognition with a Syllable-to-Character Converter","date":"2020-11-17","arxiv_id":"2011.08469","repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-automatic-speech-recognition-system","title":"Refining Automatic Speech Recognition System for older adults","date":"2020-11-17","arxiv_id":"2011.08346","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-multi-channel-integration-and","title":"Audio-visual Multi-channel Integration and Recognition of Overlapped Speech","date":"2020-11-16","arxiv_id":"2011.07755","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-shallow-fusion-for-rnn-t-personalization","title":"Deep Shallow Fusion for RNN-T Personalization","date":"2020-11-16","arxiv_id":"2011.07754","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-guided-by-contextual","title":"Improving Speech Enhancement Performance by Leveraging Contextual Broad Phonetic Class Information","date":"2020-11-15","arxiv_id":"2011.07442","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-reinforcement-learning-for","title":"Self-supervised reinforcement learning for speaker localisation with the iCub humanoid robot","date":"2020-11-12","arxiv_id":"2011.06544","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-speech-to-speech-translation","title":"Simultaneous Speech-to-Speech Translation System with Neural Incremental ASR, MT, and TTS","date":"2020-11-10","arxiv_id":"2011.04845","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-lf-mmi-ctc-and-rnn-t-criteria","title":"Benchmarking LF-MMI, CTC and RNN-T Criteria for Streaming ASR","date":"2020-11-09","arxiv_id":"2011.04785","repositories_listed":0,"syntology":null},{"url":null,"slug":"gated-recurrent-fusion-with-joint-training","title":"Gated Recurrent Fusion with Joint Training Framework for Robust End-to-End Speech Recognition","date":"2020-11-09","arxiv_id":"2011.04249","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-query-rewriting-in","title":"Personalized Query Rewriting in Conversational AI Agents","date":"2020-11-09","arxiv_id":"2011.04748","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-usefulness-of-self-attention-for","title":"On the Usefulness of Self-Attention for Automatic Speech Recognition with Transformers","date":"2020-11-08","arxiv_id":"2011.04906","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustics-based-intent-recognition-using","title":"Acoustics Based Intent Recognition Using Discovered Phonetic Units for Low Resource Languages","date":"2020-11-07","arxiv_id":"2011.03646","repositories_listed":0,"syntology":null},{"url":null,"slug":"espnet-se-end-to-end-speech-enhancement-and","title":"ESPnet-se: end-to-end speech enhancement and separation toolkit designed for asr integration","date":"2020-11-07","arxiv_id":"2011.03706","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-restricted-streaming-recurrent","title":"Alignment Restricted Streaming Recurrent Neural Network Transducer","date":"2020-11-05","arxiv_id":"2011.03072","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-images-for-asr-and-tts-through","title":"Augmenting Images for ASR and TTS through Single-loop and Dual-loop Multimodal Chain Framework","date":"2020-11-04","arxiv_id":"2011.02099","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-end-to-end-code","title":"Data Augmentation for End-to-end Code-switching Speech Recognition","date":"2020-11-04","arxiv_id":"2011.02160","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-machine-speech-chain-towards","title":"Incremental Machine Speech Chain Towards Enabling Listening while Speaking in Real-time","date":"2020-11-04","arxiv_id":"2011.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-to-sequence-learning-via-attention","title":"Sequence-to-Sequence Learning via Attention Transfer for Incremental Speech Recognition","date":"2020-11-04","arxiv_id":"2011.02127","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-transducer-with-normalized","title":"Improving RNN transducer with normalized jointer network","date":"2020-11-03","arxiv_id":"2011.01576","repositories_listed":0,"syntology":null},{"url":null,"slug":"integration-of-speech-separation-diarization","title":"Integration of speech separation, diarization, and recognition for multi-speaker meetings: System description, comparison, and analysis","date":"2020-11-03","arxiv_id":"2011.02014","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-language-model-estimation-for-domain","title":"Internal Language Model Estimation for Domain-Adaptive End-to-End Speech Recognition","date":"2020-11-03","arxiv_id":"2011.01991","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-attention-based-models-with","title":"Streaming Attention-Based Models with Augmented Memory for End-to-End Speech Recognition","date":"2020-11-03","arxiv_id":"2011.07120","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-pattern-discovery-from-thematic","title":"Unsupervised Pattern Discovery from Thematic Speech Archives Based on Multilingual Bottleneck Features","date":"2020-11-03","arxiv_id":"2011.01986","repositories_listed":0,"syntology":null},{"url":null,"slug":"warped-language-models-for-noise-robust","title":"Warped Language Models for Noise Robust Language Understanding","date":"2020-11-03","arxiv_id":"2011.01900","repositories_listed":0,"syntology":null},{"url":null,"slug":"dnn-based-semantic-model-for-rescoring-n-best","title":"DNN-Based Semantic Model for Rescoring N-best Speech Recognition List","date":"2020-11-02","arxiv_id":"2011.00975","repositories_listed":0,"syntology":null},{"url":null,"slug":"sapaugment-learning-a-sample-adaptive-policy","title":"SapAugment: Learning A Sample Adaptive Policy for Data Augmentation","date":"2020-11-02","arxiv_id":"2011.01156","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectively-pretraining-a-speech-translation","title":"Effectively pretraining a speech translation decoder with Machine Translation data","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"elitr-european-live-translator","title":"ELITR: European Live Translator","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-asr-on-alzheimers-disease-detection-1","title":"Impact of ASR on Alzheimer’s Disease Detection: All Errors are Equal, but Deletions are More Equal than Others","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-end-to-end-bangla-speech","title":"Improving End-to-End Bangla Speech Recognition with Semi-supervised Training","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"directional-asr-a-new-paradigm-for-e2e-multi","title":"Directional ASR: A New Paradigm for E2E Multi-Speaker Speech Recognition with Source Localization","date":"2020-10-30","arxiv_id":"2011.00091","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-simultaneous-speech-translation","title":"Streaming Simultaneous Speech Translation with Augmented Memory Transformer","date":"2020-10-30","arxiv_id":"2011.00033","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-speech-recognition-via-graph","title":"Semi-Supervised Speech Recognition via Graph-based Temporal Classification","date":"2020-10-29","arxiv_id":"2010.15653","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupling-pronunciation-and-language-for-end","title":"Decoupling Pronunciation and Language for End-to-end Code-switching Automatic Speech Recognition","date":"2020-10-28","arxiv_id":"2010.14798","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusion-models-for-improved-visual-captioning","title":"Fusion Models for Improved Visual Captioning","date":"2020-10-28","arxiv_id":"2010.15251","repositories_listed":0,"syntology":null},{"url":null,"slug":"int8-winograd-acceleration-for-conv1d","title":"INT8 Winograd Acceleration for Conv1D Equipped ASR Models Deployed on Mobile Devices","date":"2020-10-28","arxiv_id":"2010.14841","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-transformer-asr-with-ctc","title":"Non-Autoregressive Transformer ASR with CTC-Enhanced Decoder Input","date":"2020-10-28","arxiv_id":"2010.15025","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-encoders-for-unifying-streaming-and","title":"Cascaded encoders for unifying streaming and non-streaming ASR","date":"2020-10-27","arxiv_id":"2010.14606","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-decoder-masking-for-transformer","title":"Effective Decoder Masking for Transformer Based End-to-End Speech Recognition","date":"2020-10-27","arxiv_id":"2010.14764","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotion-recognition-by-fusing-time","title":"Emotion recognition by fusing time synchronous and time asynchronous representations","date":"2020-10-27","arxiv_id":"2010.14102","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-mask-ctc-for-non-autoregressive-end","title":"Improved Mask-CTC for Non-Autoregressive End-to-End ASR","date":"2020-10-26","arxiv_id":"2010.13270","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-noise-robustness-of-an-end-to-end","title":"Improving Noise Robustness of an End-to-End Neural Model for Automatic Speech Recognition","date":"2020-10-23","arxiv_id":"2010.12715","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-streaming-automatic-speech","title":"Improving Streaming Automatic Speech Recognition With Non-Streaming Model Distillation On Unsupervised Data","date":"2020-10-22","arxiv_id":"2010.12096","repositories_listed":0,"syntology":null},{"url":null,"slug":"mam-masked-acoustic-modeling-for-end-to-end","title":"MAM: Masked Acoustic Modeling for End-to-End Speech-to-Text Translation","date":"2020-10-22","arxiv_id":"2010.11445","repositories_listed":0,"syntology":null},{"url":null,"slug":"slimipl-language-model-free-iterative-pseudo","title":"SlimIPL: Language-Model-Free Iterative Pseudo-Labeling","date":"2020-10-22","arxiv_id":"2010.11524","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-multi-task-learning-framework-to","title":"A General Multi-Task Learning Framework to Leverage Text Data for Speech to Text Tasks","date":"2020-10-21","arxiv_id":"2010.11338","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-models-with-cyclic-feedback-for","title":"Cascaded Models With Cyclic Feedback For Direct Speech Translation","date":"2020-10-21","arxiv_id":"2010.11153","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-improved-accuracy","title":"Knowledge Distillation for Improved Accuracy in Spoken Question Answering","date":"2020-10-21","arxiv_id":"2010.11067","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentence-boundary-augmentation-for-neural","title":"Sentence Boundary Augmentation For Neural Machine Translation Robustness","date":"2020-10-21","arxiv_id":"2010.11132","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-transfer-for-efficient-on-device","title":"Knowledge Transfer for Efficient On-device False Trigger Mitigation","date":"2020-10-20","arxiv_id":"2010.10591","repositories_listed":0,"syntology":null},{"url":null,"slug":"replacing-human-audio-with-synthetic-audio","title":"Replacing Human Audio with Synthetic Audio for On-device Unspoken Punctuation Prediction","date":"2020-10-20","arxiv_id":"2010.10203","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-chinese-end-to-end-spoken-language","title":"Ensemble Chinese End-to-End Spoken Language Understanding for Abnormal Event Detection from audio stream","date":"2020-10-19","arxiv_id":"2010.09235","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-data-distillation-for-end-to-end-1","title":"Towards Data Distillation for End-to-end Spoken Conversational Question Answering","date":"2020-10-18","arxiv_id":"2010.08923","repositories_listed":0,"syntology":null},{"url":null,"slug":"studying-the-similarity-of-covid-19-sounds","title":"Studying the Similarity of COVID-19 Sounds based on Correlation Analysis of MFCC","date":"2020-10-17","arxiv_id":"2010.08770","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-speech-recognition-with","title":"Multimodal Speech Recognition with Unstructured Audio Masking","date":"2020-10-16","arxiv_id":"2010.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-intrusive-speech-intelligibility","title":"Non-intrusive speech intelligibility prediction using automatic speech recognition derived measures","date":"2020-10-16","arxiv_id":"2010.08574","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-end-to-end-speech-recognition","title":"Lightweight End-to-End Speech Recognition from Raw Audio Data Using Sinc-Convolutions","date":"2020-10-15","arxiv_id":"2010.07597","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-spectral-augmentation-for-code","title":"Exploiting Spectral Augmentation for Code-Switched Spoken Language Identification","date":"2020-10-14","arxiv_id":"2010.07130","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-low-resource-code-switched-asr","title":"Improving Low Resource Code-switched ASR using Augmented Code-switched TTS","date":"2020-10-12","arxiv_id":"2010.05549","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-asr-unify-and-improve-streaming-asr-1","title":"Dual-mode ASR: Unify and Improve Streaming ASR with Full-context Modeling","date":"2020-10-12","arxiv_id":"2010.06030","repositories_listed":0,"syntology":null},{"url":null,"slug":"wer-we-are-and-wer-we-think-we-are","title":"WER we are and WER we think we are","date":"2020-10-07","arxiv_id":"2010.03432","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-lip-localization-techniques-used","title":"A Study on Lip Localization Techniques used for Lip reading from a Video","date":"2020-09-28","arxiv_id":"2009.13420","repositories_listed":0,"syntology":null},{"url":null,"slug":"fluentnet-end-to-end-detection-of-speech","title":"FluentNet: End-to-End Detection of Speech Disfluency with Deep Learning","date":"2020-09-23","arxiv_id":"2009.11394","repositories_listed":0,"syntology":null},{"url":null,"slug":"easyasr-a-distributed-machine-learning","title":"EasyASR: A Distributed Machine Learning Platform for End-to-end Automatic Speech Recognition","date":"2020-09-14","arxiv_id":"2009.06487","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-embeddings-using-multi-task","title":"Multi-modal embeddings using multi-task learning for emotion recognition","date":"2020-09-10","arxiv_id":"2009.05019","repositories_listed":0,"syntology":null},{"url":null,"slug":"unmanned-aerial-vehicle-control-through","title":"Unmanned Aerial Vehicle Control Through Domain-based Automatic Speech Recognition","date":"2020-09-09","arxiv_id":"2009.04215","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-spoken-language-understanding-with-rl","title":"Robust Spoken Language Understanding with RL-based Value Error Recovery","date":"2020-09-07","arxiv_id":"2009.03095","repositories_listed":0,"syntology":null},{"url":null,"slug":"silent-speech-interfaces-for-speech","title":"Silent Speech Interfaces for Speech Restoration: A Review","date":"2020-09-04","arxiv_id":"2009.02110","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-by-cascading-automatic","title":"Voice Conversion by Cascading Automatic Speech Recognition and Text-to-Speech Synthesis with Prosody Transfer","date":"2020-09-03","arxiv_id":"2009.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"convolutional-speech-recognition-with-pitch","title":"Convolutional Speech Recognition with Pitch and Voice Quality Features","date":"2020-09-02","arxiv_id":"2009.01309","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-attention-based-speech-enhancement","title":"Multi-view Attention-based Speech Enhancement Model for Noise-robust Automatic Speech Recognition","date":"2020-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"aphasic-speech-recognition-using-a-mixture-of","title":"Aphasic Speech Recognition using a Mixture of Speech Intelligibility Experts","date":"2020-08-25","arxiv_id":"2008.10788","repositories_listed":0,"syntology":null},{"url":null,"slug":"learned-transferable-architectures-can","title":"Learned Transferable Architectures Can Surpass Hand-Designed Architectures for Large Scale Speech Recognition","date":"2020-08-25","arxiv_id":"2008.11589","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-tail-performance-of-a-deliberation","title":"Improving Tail Performance of a Deliberation E2E ASR Model Using a Large Text Corpus","date":"2020-08-24","arxiv_id":"2008.10491","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-utterance-language-models-with-acoustic","title":"Cross-Utterance Language Models with Acoustic Error Sampling","date":"2020-08-19","arxiv_id":"2009.01008","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-semantics-improve-asr-and-nlu","title":"Speech To Semantics: Improve ASR and NLU Jointly via All-Neural Interfaces","date":"2020-08-14","arxiv_id":"2008.06173","repositories_listed":0,"syntology":null},{"url":null,"slug":"conv-transformer-transducer-low-latency-low","title":"Conv-Transformer Transducer: Low Latency, Low Frame Rate, Streamable End-to-End Speech Recognition","date":"2020-08-13","arxiv_id":"2008.05750","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-transfer-learning-for-low","title":"Large-scale Transfer Learning for Low-resource Spoken Language Understanding","date":"2020-08-13","arxiv_id":"2008.05671","repositories_listed":0,"syntology":null},{"url":"/paper/masri-headset-a-maltese-corpus-for-speech-1","slug":"masri-headset-a-maltese-corpus-for-speech-1","title":"MASRI-HEADSET: A Maltese Corpus for Speech Recognition","date":"2020-08-13","arxiv_id":"2008.05760","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-automatic-speech-recognition-with","title":"Online Automatic Speech Recognition with Listen, Attend and Spell Model","date":"2020-08-12","arxiv_id":"2008.05514","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-approaches-for-streaming","title":"Transfer Learning Approaches for Streaming End-to-End Speech Recognition System","date":"2020-08-12","arxiv_id":"2008.05086","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-with-bidirectional-decoder-for","title":"Transformer with Bidirectional Decoder for Speech Recognition","date":"2020-08-11","arxiv_id":"2008.04481","repositories_listed":0,"syntology":null},{"url":null,"slug":"subword-regularization-an-analysis-of","title":"Subword Regularization: An Analysis of Scalability and Generalization for End-to-End Automatic Speech Recognition","date":"2020-08-10","arxiv_id":"2008.04034","repositories_listed":0,"syntology":null},{"url":null,"slug":"lrspeech-extremely-low-resource-speech","title":"LRSpeech: Extremely Low-Resource Speech Synthesis and Recognition","date":"2020-08-09","arxiv_id":"2008.03687","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-dereverberation-of","title":"Deep Learning Based Dereverberation of Temporal Envelopesfor Robust Speech Recognition","date":"2020-08-07","arxiv_id":"2008.03339","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-speaker-adaptation-methods","title":"Investigation of Speaker-adaptation methods in Transformer based ASR","date":"2020-08-07","arxiv_id":"2008.03247","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-transfer-learning-method-for-speech-emotion","title":"A Transfer Learning Method for Speech Emotion Recognition from Automatic Speech Recognition","date":"2020-08-06","arxiv_id":"2008.02863","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-compression-of-end-to-end-asr-model","title":"Iterative Compression of End-to-End ASR Model using AutoML","date":"2020-08-06","arxiv_id":"2008.02897","repositories_listed":0,"syntology":null},{"url":null,"slug":"shouted-speech-compensation-for-speaker","title":"Shouted Speech Compensation for Speaker Verification Robust to Vocal Effort Conditions","date":"2020-08-06","arxiv_id":"2008.02487","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-cross-domain-singing-voice","title":"Unsupervised Cross-Domain Singing Voice Conversion","date":"2020-08-06","arxiv_id":"2008.02830","repositories_listed":0,"syntology":null},{"url":null,"slug":"this-is-houston-say-again-please-the-behavox","title":"\"This is Houston. Say again, please\". The Behavox system for the Apollo-11 Fearless Steps Challenge (phase II)","date":"2020-08-04","arxiv_id":"2008.01504","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-construction-of-asr-systems","title":"Weakly Supervised Construction of ASR Systems with Massive Video Data","date":"2020-08-04","arxiv_id":"2008.01300","repositories_listed":0,"syntology":null}],"record_sha256":"0c54944fb72329c6251f3f7395a7dd9b969d6da2dcd6ab1d18cbb787c6b416e7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}