{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/23","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":23,"pages_in_order":32,"rows_per_page":100,"rows":[2201,2300],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/22","next":"/task/automatic-speech-recognition-2/papers/24","papers":[{"url":null,"slug":"cross-modal-transformer-based-neural","title":"Cross-Modal Transformer-Based Neural Correction Models for Automatic Speech Recognition","date":"2021-07-04","arxiv_id":"2107.01569","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-autoregressive-modeling-for-joint-end","title":"Unified Autoregressive Modeling for Joint End-to-End Multi-Talker Overlapped Speech Recognition and Speaker Attribute Estimation","date":"2021-07-04","arxiv_id":"2107.01549","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-causal-non-causal-self-attention-for","title":"Dual Causal/Non-Causal Self-Attention for Streaming End-to-End Speech Recognition","date":"2021-07-02","arxiv_id":"2107.01269","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-user-voicefilter-lite-via-attentive","title":"Multi-user VoiceFilter-Lite via Attentive Speaker Embedding","date":"2021-07-02","arxiv_id":"2107.01201","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-named-entity-recognition-in-spoken","title":"Improving Named Entity Recognition in Spoken Dialog Systems by Context and Speech Pattern Modeling","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"smarterp-a-cai-system-to-support-simultaneous","title":"SmarTerp: A CAI System to Support Simultaneous Interpreters in Real-Time","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stableemit-selection-probability-discount-for","title":"StableEmit: Selection Probability Discount for Reducing Emission Latency of Streaming Monotonic Attention ASR","date":"2021-07-01","arxiv_id":"2107.00635","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-free-spoken-language-understanding-for","title":"Word-Free Spoken Language Understanding for Mandarin-Chinese","date":"2021-07-01","arxiv_id":"2107.00186","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-spoken-language-understanding-2","title":"On joint training with interfaces for spoken language understanding","date":"2021-06-30","arxiv_id":"2106.15919","repositories_listed":0,"syntology":null},{"url":null,"slug":"ims-systems-for-the-iwslt-2021-low-resource","title":"IMS' Systems for the IWSLT 2021 Low-Resource Speech Translation Task","date":"2021-06-30","arxiv_id":"2106.16055","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-confidence-classifier-for-asr","title":"Sequence-level Confidence Classifier for ASR Utterance Accuracy and Application to Acoustic Models","date":"2021-06-30","arxiv_id":"2107.00099","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-end-to-end-evaluation-of","title":"Rethinking End-to-End Evaluation of Decomposable Tasks: A Case Study on Spoken Language Understanding","date":"2021-06-29","arxiv_id":"2106.15065","repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large","title":"QASR: QCRI Aljazeera Speech Resource -- A Large Scale Annotated Arabic Speech Corpus","date":"2021-06-24","arxiv_id":"2106.13000","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-are-we-in-semantic-concept-extraction","title":"Where are we in semantic concept extraction for Spoken Language Understanding?","date":"2021-06-24","arxiv_id":"2106.13045","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixtures-of-deep-neural-experts-for-automated","title":"Mixtures of Deep Neural Experts for Automated Speech Scoring","date":"2021-06-23","arxiv_id":"2106.12475","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-joint-modeling-of-multiple-spoken","title":"Zero-Shot Joint Modeling of Multiple Spoken-Text-Style Conversion Tasks using Switching Tokens","date":"2021-06-23","arxiv_id":"2106.12131","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-discriminative-entity-aware-language-model","title":"A Discriminative Entity-Aware Language Model for Virtual Assistants","date":"2021-06-21","arxiv_id":"2106.11292","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-single-step-non-autoregressive","title":"An Improved Single Step Non-autoregressive Transformer for Automatic Speech Recognition","date":"2021-06-18","arxiv_id":"2106.09885","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-german-asr-with-untranscribed","title":"Low Resource German ASR with Untranscribed Data Spoken by Non-native Children -- INTERSPEECH 2021 Shared Task SPAPL System","date":"2021-06-18","arxiv_id":"2106.09963","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-personalization-of-automatic-speech","title":"On-Device Personalization of Automatic Speech Recognition Models for Disordered Speech","date":"2021-06-18","arxiv_id":"2106.10259","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-pruning-on-demand-with-intermediate-ctc","title":"Layer Pruning on Demand with Intermediate CTC","date":"2021-06-17","arxiv_id":"2106.09216","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-mode-transformer-transducer-with","title":"Multi-mode Transformer Transducer with Stochastic Future Context","date":"2021-06-17","arxiv_id":"2106.09760","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-classification-on-spoken-documents","title":"Topic Classification on Spoken Documents Using Deep Acoustic and Linguistic Features","date":"2021-06-16","arxiv_id":"2106.08637","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-into-pre-training-strategies-for","title":"A Study into Pre-training Strategies for Spoken Language Understanding on Dysarthric Speech","date":"2021-06-15","arxiv_id":"2106.08313","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-adaptation-for-e-commerce-chatbots-using","title":"ASR Adaptation for E-commerce Chatbots using Cross-Utterance Context and Multi-Task Language Modeling","date":"2021-06-15","arxiv_id":"2106.09532","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialectal-speech-recognition-and-translation","title":"Dialectal Speech Recognition and Translation of Swiss German Speech to Standard German Text: Microsoft's Submission to SwissText 2021","date":"2021-06-15","arxiv_id":"2106.08126","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-opus-compression-for-far-field","title":"Multi-channel Opus compression for far-field automatic speech recognition with a fixed bitrate budget","date":"2021-06-15","arxiv_id":"2106.07994","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-domain-mismatch-in-low-resource","title":"Overcoming Domain Mismatch in Low Resource Sequence-to-Sequence ASR Models using Hybrid Generated Pseudotranscripts","date":"2021-06-14","arxiv_id":"2106.07716","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthasr-unlocking-synthetic-data-for-speech","title":"SynthASR: Unlocking Synthetic Data for Speech Recognition","date":"2021-06-14","arxiv_id":"2106.07803","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-heterogeneity-in-semi-supervised","title":"Using heterogeneity in semi-supervised transcription hypotheses to improve code-switched speech recognition","date":"2021-06-14","arxiv_id":"2106.07699","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-sentence-neural-language-models-for","title":"Cross-utterance Reranking Models with BERT and Graph Convolutional Networks for Conversational Speech Recognition","date":"2021-06-13","arxiv_id":"2106.06922","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-t-asr-performance-with-date","title":"Improving RNN-T ASR Performance with Date-Time and Location Awareness","date":"2021-06-11","arxiv_id":"2106.06183","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-pre-trained-language-model-for","title":"Leveraging Pre-trained Language Model for Speech Sentiment Analysis","date":"2021-06-11","arxiv_id":"2106.06598","repositories_listed":0,"syntology":null},{"url":null,"slug":"parp-prune-adjust-and-re-prune-for-self","title":"PARP: Prune, Adjust and Re-Prune for Self-Supervised Speech Recognition","date":"2021-06-10","arxiv_id":"2106.05933","repositories_listed":0,"syntology":null},{"url":"/paper/task-aware-multi-task-learning-for-speech-to","slug":"task-aware-multi-task-learning-for-speech-to","title":"TASK AWARE MULTI-TASK LEARNING FOR SPEECH TO TEXT TASKS","date":"2021-06-10","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-neural-architectures","title":"A Comparative Study on Neural Architectures and Training Methods for Japanese Speech Recognition","date":"2021-06-09","arxiv_id":"2106.05111","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-automatic-speech-recognition-a","title":"Unsupervised Automatic Speech Recognition: A Review","date":"2021-06-09","arxiv_id":"2106.04897","repositories_listed":0,"syntology":null},{"url":"/paper/sequential-end-to-end-intent-and-slot-label","slug":"sequential-end-to-end-intent-and-slot-label","title":"Sequential End-to-End Intent and Slot Label Classification and Localization","date":"2021-06-08","arxiv_id":"2106.04660","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-methods-for-end-to-end","title":"Data Augmentation Methods for End-to-end Speech Recognition on Distant-Talk Scenarios","date":"2021-06-07","arxiv_id":"2106.03419","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-listening-and-live-captioning-multi","title":"Human Listening and Live Captioning: Multi-Task Training for Speech Enhancement","date":"2021-06-05","arxiv_id":"2106.02896","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-you-listen-with-one-or-two-microphones-a","title":"Do You Listen with One or Two Microphones? A Unified ASR Model for Single and Multi-Channel Audio","date":"2021-06-04","arxiv_id":"2106.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-wer-a-unified-metric-for-the","title":"Semantic-WER: A Unified Metric for the Evaluation of ASR Transcript for End Usability","date":"2021-06-03","arxiv_id":"2106.02016","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-script-e2e-framework-for-multilingual","title":"Dual Script E2E framework for Multilingual and Code-Switching ASR","date":"2021-06-02","arxiv_id":"2106.01400","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-low-resource-asr-performance-with","title":"Improving low-resource ASR performance with untranscribed out-of-domain data","date":"2021-06-02","arxiv_id":"2106.01227","repositories_listed":0,"syntology":null},{"url":null,"slug":"should-we-always-separate-switching-between","title":"Should We Always Separate?: Switching Between Enhanced and Observed Signals for Overlapping Speech Recognition","date":"2021-06-02","arxiv_id":"2106.00949","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-neural-acoustic-echo-canceller-optimized","title":"A Neural Acoustic Echo Canceller Optimized Using An Automatic Speech Recognizer And Large Scale Synthetic Data","date":"2021-06-01","arxiv_id":"2106.00856","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-asr-for-indonesian-english","title":"Developing ASR for Indonesian-English Bilingual Language Teaching","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-asr-to-jointly-predict","title":"End-to-end ASR to jointly predict transcriptions and linguistic annotations","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-automatic-speech-recognition-its","title":"End-to-End Automatic Speech Recognition: Its Impact on the Workflowin Documenting Yoloxóchitl Mixtec","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-automatic-speech-recognition-1","title":"Evaluating Automatic Speech Recognition Quality and Its Impact on Counselor Utterance Coding","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"highland-puebla-nahuatl-speech-translation","title":"Highland Puebla Nahuatl Speech Translation Corpus for Endangered Language Documentation","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-one-model-to-rule-all-multilingual","title":"Towards One Model to Rule All: Multilingual Strategy for Dialectal Code-Switching Arabic ASR","date":"2021-05-31","arxiv_id":"2105.14779","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-speech-enhancement-systems-with","title":"Training Speech Enhancement Systems with Noisy Speech Datasets","date":"2021-05-26","arxiv_id":"2105.12315","repositories_listed":0,"syntology":null},{"url":null,"slug":"mondegreen-a-post-processing-solution-to","title":"Mondegreen: A Post-Processing Solution to Speech Recognition Error Correction for Voice Search Queries","date":"2021-05-20","arxiv_id":"2105.09930","repositories_listed":0,"syntology":null},{"url":null,"slug":"listra-automatic-speech-translation-english","title":"LiSTra, Automatic Speech Translation: English to Lingala case study","date":"2021-05-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"listen-with-intent-improving-speech","title":"Listen with Intent: Improving Speech Recognition with Audio-to-Intent Front-End","date":"2021-05-14","arxiv_id":"2105.07071","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-transformer-for-hardware-efficient","title":"Streaming Transformer for Hardware Efficient Voice Trigger Detection and False Trigger Mitigation","date":"2021-05-14","arxiv_id":"2105.06598","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-ctc-based-end-to-end-techniques-for","title":"Exploring CTC Based End-to-End Techniques for Myanmar Speech Recognition","date":"2021-05-13","arxiv_id":"2105.06253","repositories_listed":0,"syntology":null},{"url":null,"slug":"stacked-acoustic-and-textual-encoding","title":"Stacked Acoustic-and-Textual Encoding: Integrating the Pre-trained Models into Speech Translation Encoders","date":"2021-05-12","arxiv_id":"2105.05752","repositories_listed":0,"syntology":null},{"url":null,"slug":"stutternet-stuttering-detection-using-time","title":"StutterNet: Stuttering Detection Using Time Delay Neural Network","date":"2021-05-12","arxiv_id":"2105.05599","repositories_listed":0,"syntology":null},{"url":"/paper/speech2slot-an-end-to-end-knowledge-based","slug":"speech2slot-an-end-to-end-knowledge-based","title":"Speech2Slot: An End-to-End Knowledge-based Slot Filling from Speech","date":"2021-05-10","arxiv_id":"2105.04719","repositories_listed":0,"syntology":null},{"url":null,"slug":"english-accent-accuracy-analysis-in-a-state","title":"English Accent Accuracy Analysis in a State-of-the-Art Automatic Speech Recognition System","date":"2021-05-09","arxiv_id":"2105.05041","repositories_listed":0,"syntology":null},{"url":null,"slug":"latency-controlled-neural-architecture-search","title":"Latency-Controlled Neural Architecture Search for Streaming Speech Recognition","date":"2021-05-08","arxiv_id":"2105.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-end-to-end-automatic-speech","title":"Robustness of end-to-end Automatic Speech Recognition Models -- A Case Study using Mozilla DeepSpeech","date":"2021-05-08","arxiv_id":"2105.09742","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-recognition-with-hybrid-phonetic","title":"Accent Recognition with Hybrid Phonetic Features","date":"2021-05-05","arxiv_id":"2105.01920","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-modification-for-recognition-of","title":"Spectral modification for recognition of children’s speech undermismatched conditions","date":"2021-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-keyphrase-detection-using","title":"Personalized Keyphrase Detection using Speaker and Environment Information","date":"2021-04-28","arxiv_id":"2104.13970","repositories_listed":0,"syntology":null},{"url":null,"slug":"head-synchronous-decoding-for-transformer","title":"Head-synchronous Decoding for Transformer-based Streaming ASR","date":"2021-04-26","arxiv_id":"2104.12631","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-for-end-to-end-asr-word","title":"Multi-Task Learning for End-to-End ASR Word and Utterance Confidence with Deletion Prediction","date":"2021-04-26","arxiv_id":"2104.12870","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-data-augmentation-for-end-to-end","title":"Semantic Data Augmentation for End-to-End Mandarin Speech Recognition","date":"2021-04-26","arxiv_id":"2104.12521","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-streaming-and-non","title":"Bridging the gap between streaming and non-streaming ASR systems bydistilling ensembles of CTC and RNN-T models","date":"2021-04-25","arxiv_id":"2104.14346","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-of-deep-neural-networks-for-1","title":"Quantization of Deep Neural Networks for Accurate Edge Computing","date":"2021-04-25","arxiv_id":"2104.12046","repositories_listed":0,"syntology":null},{"url":null,"slug":"accented-speech-recognition-a-survey","title":"Accented Speech Recognition: A Survey","date":"2021-04-21","arxiv_id":"2104.10747","repositories_listed":0,"syntology":null},{"url":null,"slug":"discriminative-self-training-for-punctuation","title":"Discriminative Self-training for Punctuation Prediction","date":"2021-04-21","arxiv_id":"2104.10339","repositories_listed":0,"syntology":null},{"url":null,"slug":"disfluency-detection-with-unlabeled-data-and","title":"Disfluency Detection with Unlabeled Data and Small BERT Models","date":"2021-04-21","arxiv_id":"2104.10769","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-synchronous-speech-to-text-alignment","title":"Label-Synchronous Speech-to-Text Alignment for ASR Using Forward and Backward Transformers","date":"2021-04-21","arxiv_id":"2104.10328","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-sampling-based-training-criteria-for","title":"On Sampling-Based Training Criteria for Neural Language Modeling","date":"2021-04-21","arxiv_id":"2104.10507","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-for-spoken-language","title":"Pre-training for Spoken Language Understanding with Joint Textual and Phonetic Representation Learning","date":"2021-04-21","arxiv_id":"2104.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-aware-far-field-automatic-speech","title":"Scene-aware Far-field Automatic Speech Recognition","date":"2021-04-21","arxiv_id":"2104.10757","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-impact-of-word-error-rate-on-acoustic","title":"On the Impact of Word Error Rate on Acoustic-Linguistic Speech Emotion Recognition: An Update for the Deep Learning Era","date":"2021-04-20","arxiv_id":"2104.10121","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-data-driven-subword-modeling-for-end","title":"Acoustic Data-Driven Subword Modeling for End-to-End Speech Recognition","date":"2021-04-19","arxiv_id":"2104.09106","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-long-context-end-to-end-speech","title":"Advanced Long-context End-to-end Speech Recognition Using Context-expanded Transformers","date":"2021-04-19","arxiv_id":"2104.09426","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-practices-for-noise-based-augmentation","title":"Best Practices for Noise-Based Augmentation to Improve the Performance of Deployable Speech-Based Emotion Recognition Systems","date":"2021-04-18","arxiv_id":"2104.08806","repositories_listed":0,"syntology":null},{"url":null,"slug":"mimo-self-attentive-rnn-beamformer-for-multi","title":"MIMO Self-attentive RNN Beamformer for Multi-speaker Speech Separation","date":"2021-04-17","arxiv_id":"2104.08450","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-clean-data-training","title":"Bridging the Gap Between Clean Data Training and Real-World Inference for Spoken Language Understanding","date":"2021-04-13","arxiv_id":"2104.06393","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivalence-of-segmental-and-neural","title":"Equivalence of Segmental and Neural Transducer Modeling: A Proof of Concept","date":"2021-04-13","arxiv_id":"2104.06104","repositories_listed":0,"syntology":null},{"url":null,"slug":"source-and-target-bidirectional-knowledge","title":"Source and Target Bidirectional Knowledge Distillation for End-to-end Speech Translation","date":"2021-04-13","arxiv_id":"2104.06457","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-the-benefit-of-synthetic-training","title":"Comparing the Benefit of Synthetic Training Data for Various Automatic Speech Recognition Architectures","date":"2021-04-12","arxiv_id":"2104.05379","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-conformer-based-end-to-end-speech","title":"Improved Conformer-based End-to-End Speech Recognition Using Neural Architecture Search","date":"2021-04-12","arxiv_id":"2104.05390","repositories_listed":0,"syntology":null},{"url":null,"slug":"innovative-bert-based-reranking-language","title":"Innovative Bert-based Reranking Language Models for Speech Recognition","date":"2021-04-11","arxiv_id":"2104.04950","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-transformer-based-end-to","title":"Non-autoregressive Transformer-based End-to-end ASR using BERT","date":"2021-04-10","arxiv_id":"2104.04805","repositories_listed":0,"syntology":null},{"url":null,"slug":"accented-speech-recognition-inspired-by-human","title":"Accented Speech Recognition Inspired by Human Perception","date":"2021-04-09","arxiv_id":"2104.04627","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-replacement-and-combination-for","title":"On Architectures and Training for Raw Waveform Feature Extraction in ASR","date":"2021-04-09","arxiv_id":"2104.04298","repositories_listed":0,"syntology":null},{"url":"/paper/bstc-a-large-scale-chinese-english-speech","slug":"bstc-a-large-scale-chinese-english-speech","title":"BSTC: A Large-Scale Chinese-English Speech Translation Dataset","date":"2021-04-08","arxiv_id":"2104.03575","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-semi-supervised-learning-an","title":"Contextual Semi-Supervised Learning: An Approach To Leverage Air-Surveillance and Untranscribed ATC Data in ASR Systems","date":"2021-04-08","arxiv_id":"2104.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-machine-speech-chain-for-domain","title":"Exploring Machine Speech Chain for Domain Adaptation and Few-Shot Speaker Adaptation","date":"2021-04-08","arxiv_id":"2104.03815","repositories_listed":0,"syntology":null},{"url":null,"slug":"wnars-wfst-based-non-autoregressive-streaming","title":"WNARS: WFST based Non-autoregressive Streaming End-to-End Speech Recognition","date":"2021-04-08","arxiv_id":"2104.03587","repositories_listed":0,"syntology":null},{"url":null,"slug":"capturing-multi-resolution-context-by-dilated","title":"Capturing Multi-Resolution Context by Dilated Self-Attention","date":"2021-04-07","arxiv_id":"2104.02858","repositories_listed":0,"syntology":null},{"url":null,"slug":"pushing-the-limits-of-non-autoregressive","title":"Pushing the Limits of Non-Autoregressive Speech Recognition","date":"2021-04-07","arxiv_id":"2104.03416","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-ctc-and-lfmmi-for-out-of-domain","title":"Comparing CTC and LFMMI for out-of-domain adaptation of wav2vec 2.0 acoustic model","date":"2021-04-06","arxiv_id":"2104.02558","repositories_listed":0,"syntology":null}],"record_sha256":"87d682c795d028faf873a7c4f55652ebfc7104dbdc6ad358ada031cbe760ab9c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}