{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/37","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":37,"pages_in_order":58,"rows_per_page":100,"rows":[3601,3700],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/36","next":"/task/speech-recognition-1/papers/38","papers":[{"url":null,"slug":"a-speech-enabled-fixed-phrase-translator-for","title":"A Speech-enabled Fixed-phrase Translator for Healthcare Accessibility","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-generation-of-a-3d-sign-language","title":"Automatic generation of a 3D sign language avatar on AR glasses given 2D videos of human signers","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"avengers-ensemble-benefits-of-ensembling-in","title":"Avengers, Ensemble! Benefits of ensembling in grapheme-to-phoneme prediction","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"best-of-both-worlds-making-high-accuracy-non","title":"Best of Both Worlds: Making High Accuracy Non-incremental Transformer-based Disfluency Detection Incremental","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bts-back-transcription-for-speech-to-text","title":"BTS: Back TranScription for Speech-to-Text Post-Processor using Text-to-Speech-to-Text","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-might-we-create-better-benchmarks-for","title":"How Might We Create Better Benchmarks for Speech Recognition?","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ims-systems-for-the-iwslt-2021-low-resource-1","title":"IMS’ Systems for the IWSLT 2021 Low-Resource Speech Translation Task","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-reinforcement-learning-for-table","title":"Interactive Reinforcement Learning for Table Balancing Robot","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ji-yu-gai-jin-conformerde-xin-wen-ling-yu","title":"基于改进Conformer的新闻领域端到端语音识别(End-to-End Speech Recognition in News Field based on Conformer)","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kits-iwslt-2021-offline-speech-translation","title":"KIT’s IWSLT 2021 Offline Speech Translation System","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-speech-translation-with-unified-1","title":"Multilingual Speech Translation with Unified Transformer: Huawei Noah’s Ark Lab at IWSLT 2021","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-knowledge-distillation-for-translating","title":"On Knowledge Distillation for Translating Erroneous Speech Transcriptions","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-trac-systems-for-the-iwslt-2021-low","title":"ON-TRAC’ systems for the IWSLT 2021 low-resource speech translation and multilingual speech translation shared tasks","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large-1","title":"QASR: QCRI Aljazeera Speech Resource A Large Scale Annotated Arabic Speech Corpus","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"technology-augmented-multilingual","title":"Technology-Augmented Multilingual Communication Models: New Interaction Paradigms, Shifts in the Language Services Industry, and Implications for Training Programs","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"without-further-ado-direct-and-simultaneous","title":"Without Further Ado: Direct and Simultaneous Speech Translation by AppTek in 2021","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zjus-iwslt-2021-speech-translation-system","title":"ZJU’s IWSLT 2021 Speech Translation System","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"can-you-hear-it-backdoor-attacks-via","title":"Can You Hear It? Backdoor Attacks via Ultrasonic Triggers","date":"2021-07-30","arxiv_id":"2107.14569","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-gpt-gpt-2-and-bert-language-models","title":"Adapting GPT, GPT-2 and BERT Language Models for Speech Recognition","date":"2021-07-29","arxiv_id":"2108.07789","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-wav2vec2-an-application-of","title":"An Adapter Based Pre-Training for Efficient and Scalable Self-Supervised Speech Representation Learning","date":"2021-07-26","arxiv_id":"2107.13530","repositories_listed":0,"syntology":null},{"url":null,"slug":"facetron-multi-speaker-face-to-speech-model","title":"Facetron: A Multi-speaker Face-to-Speech Model based on Cross-modal Latent Representations","date":"2021-07-26","arxiv_id":"2107.12003","repositories_listed":0,"syntology":null},{"url":"/paper/olr-2021-challenge-datasets-rules-and","slug":"olr-2021-challenge-datasets-rules-and","title":"OLR 2021 Challenge: Datasets, Rules and Baselines","date":"2021-07-23","arxiv_id":"2107.11113","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-learning-techniques-and","title":"Using Deep Learning Techniques and Inferential Speech Statistics for AI Synthesised Speech Recognition","date":"2021-07-23","arxiv_id":"2107.11412","repositories_listed":0,"syntology":null},{"url":null,"slug":"carnelinet-neural-mixture-model-for-automatic","title":"CarneliNet: Neural Mixture Model for Automatic Speech Recognition","date":"2021-07-22","arxiv_id":"2107.10708","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitask-based-joint-learning-approach-to","title":"Multitask-Based Joint Learning Approach To Robust ASR For Radio Communication Speech","date":"2021-07-22","arxiv_id":"2107.10701","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-communications-for-speech","title":"Semantic Communications for Speech Recognition","date":"2021-07-22","arxiv_id":"2107.11190","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-prosody-modeling-for-asr-tts-based-voice","title":"On Prosody Modeling for ASR+TTS based Voice Conversion","date":"2021-07-20","arxiv_id":"2107.09477","repositories_listed":0,"syntology":null},{"url":null,"slug":"seed-words-based-data-selection-for-language","title":"Seed Words Based Data Selection for Language Model Adaptation","date":"2021-07-20","arxiv_id":"2107.09433","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-baseline-model-for-computationally","title":"A baseline model for computationally inexpensive speech recognition for Kazakh using the Coqui STT framework","date":"2021-07-19","arxiv_id":"2107.10637","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-with-cross-attention-for","title":"Multi-task Learning with Cross Attention for Keyword Spotting","date":"2021-07-15","arxiv_id":"2107.07634","repositories_listed":0,"syntology":null},{"url":null,"slug":"vad-free-streaming-hybrid-ctc-attention-asr","title":"VAD-free Streaming Hybrid CTC/Attention ASR for Unsegmented Recording","date":"2021-07-15","arxiv_id":"2107.07509","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-configurable-multilingual-model-is-all-you","title":"A Configurable Multilingual Model is All You Need to Recognize All Languages","date":"2021-07-13","arxiv_id":"2107.05876","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformer-based-end-to-end-speech-recognition","title":"Conformer-based End-to-end Speech Recognition With Rotary Position Embedding","date":"2021-07-13","arxiv_id":"2107.05907","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-iwslt-2021-but-speech-translation-systems","title":"The IWSLT 2021 BUT Speech Translation Systems","date":"2021-07-13","arxiv_id":"2107.06155","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-speech-translation","title":"Zero-shot Speech Translation","date":"2021-07-13","arxiv_id":"2107.06010","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptual-based-deep-learning-denoiser-as-a","title":"Perceptual-based deep-learning denoiser as a defense against adversarial attacks on ASR systems","date":"2021-07-12","arxiv_id":"2107.05222","repositories_listed":0,"syntology":null},{"url":null,"slug":"unispeech-at-scale-an-empirical-study-of-pre","title":"UniSpeech at scale: An Empirical Study of Pre-training Method on Large-Scale Speech Recognition Dataset","date":"2021-07-12","arxiv_id":"2107.05233","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-prediction-end-to-end-active-learning","title":"Loss Prediction: End-to-End Active Learning Approach For Speech Recognition","date":"2021-07-09","arxiv_id":"2107.04289","repositories_listed":0,"syntology":null},{"url":null,"slug":"noisy-training-improves-e2e-asr-for-the-edge","title":"Noisy Training Improves E2E ASR for the Edge","date":"2021-07-09","arxiv_id":"2107.04677","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-lattice-free-boosted-mmi-training-of-hmm","title":"On lattice-free boosted MMI training of HMM and CTC-based full-context ASR models","date":"2021-07-09","arxiv_id":"2107.04154","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-to-classify-and","title":"Representation Learning to Classify and Detect Adversarial Attacks against Speaker and Speech Recognition Systems","date":"2021-07-09","arxiv_id":"2107.04448","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-language-identification-through","title":"Improved Language Identification Through Cross-Lingual Self-Supervised Learning","date":"2021-07-08","arxiv_id":"2107.04082","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-rich-transcription-style-automatic","title":"End-to-End Rich Transcription-Style Automatic Speech Recognition with Semi-Supervised Learning","date":"2021-07-07","arxiv_id":"2107.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-accuracy-of","title":"Improving Speech Recognition Accuracy of Local POI Using Geographical Models","date":"2021-07-07","arxiv_id":"2107.03165","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-modular-and-joint","title":"A Comparative Study of Modular and Joint Approaches for Speaker-Attributed ASR on Monaural Long-Form Audio","date":"2021-07-06","arxiv_id":"2107.02852","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-single-channel-speech-for-multi","title":"Exploiting Single-Channel Speech For Multi-channel End-to-end Speech Recognition","date":"2021-07-06","arxiv_id":"2107.02670","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-a-neural-network-model-by","title":"Improving a neural network model by explanation-guided training for glioma classification based on MRI data","date":"2021-07-05","arxiv_id":"2107.02008","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-practical-aspects-of-single","title":"Investigation of Practical Aspects of Single Channel Speech Separation for ASR","date":"2021-07-05","arxiv_id":"2107.01922","repositories_listed":0,"syntology":null},{"url":null,"slug":"arabic-code-switching-speech-recognition","title":"Arabic Code-Switching Speech Recognition using Monolingual Data","date":"2021-07-04","arxiv_id":"2107.01573","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-transformer-based-neural","title":"Cross-Modal Transformer-Based Neural Correction Models for Automatic Speech Recognition","date":"2021-07-04","arxiv_id":"2107.01569","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-autoregressive-modeling-for-joint-end","title":"Unified Autoregressive Modeling for Joint End-to-End Multi-Talker Overlapped Speech Recognition and Speaker Attribute Estimation","date":"2021-07-04","arxiv_id":"2107.01549","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-causal-non-causal-self-attention-for","title":"Dual Causal/Non-Causal Self-Attention for Streaming End-to-End Speech Recognition","date":"2021-07-02","arxiv_id":"2107.01269","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-user-voicefilter-lite-via-attentive","title":"Multi-user VoiceFilter-Lite via Attentive Speaker Embedding","date":"2021-07-02","arxiv_id":"2107.01201","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-contrastive-learning-for-accented","title":"Supervised Contrastive Learning for Accented Speech Recognition","date":"2021-07-02","arxiv_id":"2107.00921","repositories_listed":0,"syntology":null},{"url":null,"slug":"espnet-st-iwslt-2021-offline-speech","title":"ESPnet-ST IWSLT 2021 Offline Speech Translation System","date":"2021-07-01","arxiv_id":"2107.00636","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-named-entity-recognition-in-spoken","title":"Improving Named Entity Recognition in Spoken Dialog Systems by Context and Speech Pattern Modeling","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-decoding-of-words-from-visual","title":"Interactive decoding of words from visual speech recognition models","date":"2021-07-01","arxiv_id":"2107.00692","repositories_listed":0,"syntology":null},{"url":null,"slug":"projection-of-turn-completion-in-incremental","title":"Projection of Turn Completion in Incremental Spoken Dialogue Systems","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"smarterp-a-cai-system-to-support-simultaneous","title":"SmarTerp: A CAI System to Support Simultaneous Interpreters in Real-Time","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"stableemit-selection-probability-discount-for","title":"StableEmit: Selection Probability Discount for Reducing Emission Latency of Streaming Monotonic Attention ASR","date":"2021-07-01","arxiv_id":"2107.00635","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-free-spoken-language-understanding-for","title":"Word-Free Spoken Language Understanding for Mandarin-Chinese","date":"2021-07-01","arxiv_id":"2107.00186","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-spoken-language-understanding-2","title":"On joint training with interfaces for spoken language understanding","date":"2021-06-30","arxiv_id":"2106.15919","repositories_listed":0,"syntology":null},{"url":null,"slug":"ims-systems-for-the-iwslt-2021-low-resource","title":"IMS' Systems for the IWSLT 2021 Low-Resource Speech Translation Task","date":"2021-06-30","arxiv_id":"2106.16055","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-confidence-classifier-for-asr","title":"Sequence-level Confidence Classifier for ASR Utterance Accuracy and Application to Acoustic Models","date":"2021-06-30","arxiv_id":"2107.00099","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-end-to-end-evaluation-of","title":"Rethinking End-to-End Evaluation of Decomposable Tasks: A Case Study on Spoken Language Understanding","date":"2021-06-29","arxiv_id":"2106.15065","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-a-novel-training-algorithm-for-sequence-to","title":"On a novel training algorithm for sequence-to-sequence predictive recurrent networks","date":"2021-06-27","arxiv_id":"2106.14120","repositories_listed":0,"syntology":null},{"url":null,"slug":"use-of-machine-learning-technique-to-maximize","title":"Use of Machine Learning Technique to maximize the signal over background for $H \\rightarrow ττ$","date":"2021-06-27","arxiv_id":"2106.14257","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-intelligent-autonomous-navigation","title":"Building Intelligent Autonomous Navigation Agents","date":"2021-06-25","arxiv_id":"2106.13415","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexical-access-model-for-italian-modeling","title":"Lexical Access Model for Italian -- Modeling human speech processing: identification of words in running speech toward lexical access based on the detection of landmarks and other acoustic cues to features","date":"2021-06-24","arxiv_id":"2107.02720","repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large","title":"QASR: QCRI Aljazeera Speech Resource -- A Large Scale Annotated Arabic Speech Corpus","date":"2021-06-24","arxiv_id":"2106.13000","repositories_listed":0,"syntology":null},{"url":null,"slug":"where-are-we-in-semantic-concept-extraction","title":"Where are we in semantic concept extraction for Spoken Language Understanding?","date":"2021-06-24","arxiv_id":"2106.13045","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixtures-of-deep-neural-experts-for-automated","title":"Mixtures of Deep Neural Experts for Automated Speech Scoring","date":"2021-06-23","arxiv_id":"2106.12475","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-joint-modeling-of-multiple-spoken","title":"Zero-Shot Joint Modeling of Multiple Spoken-Text-Style Conversion Tasks using Switching Tokens","date":"2021-06-23","arxiv_id":"2106.12131","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-discriminative-entity-aware-language-model","title":"A Discriminative Entity-Aware Language Model for Virtual Assistants","date":"2021-06-21","arxiv_id":"2106.11292","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-reach-real-time-ai-on-consumer-devices","title":"How to Reach Real-Time AI on Consumer Devices? Solutions for Programmable and Custom Architectures","date":"2021-06-21","arxiv_id":"2106.15021","repositories_listed":0,"syntology":null},{"url":null,"slug":"pay-better-attention-to-attention-head","title":"Pay Better Attention to Attention: Head Selection in Multilingual and Multi-Domain Sequence Modeling","date":"2021-06-21","arxiv_id":"2106.10840","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-the-master-distilling-cross","title":"Learning From the Master: Distilling Cross-Modal Advanced Knowledge for Lip Reading","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-single-step-non-autoregressive","title":"An Improved Single Step Non-autoregressive Transformer for Automatic Speech Recognition","date":"2021-06-18","arxiv_id":"2106.09885","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-and-tuning-of-a-voice-assistant","title":"Analysis and Tuning of a Voice Assistant System for Dysfluent Speech","date":"2021-06-18","arxiv_id":"2106.11759","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-german-asr-with-untranscribed","title":"Low Resource German ASR with Untranscribed Data Spoken by Non-native Children -- INTERSPEECH 2021 Shared Task SPAPL System","date":"2021-06-18","arxiv_id":"2106.09963","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-personalization-of-automatic-speech","title":"On-Device Personalization of Automatic Speech Recognition Models for Disordered Speech","date":"2021-06-18","arxiv_id":"2106.10259","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-pruning-on-demand-with-intermediate-ctc","title":"Layer Pruning on Demand with Intermediate CTC","date":"2021-06-17","arxiv_id":"2106.09216","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-mode-transformer-transducer-with","title":"Multi-mode Transformer Transducer with Stochastic Future Context","date":"2021-06-17","arxiv_id":"2106.09760","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-practices-for-noise-based-augmentation-1","title":"Best Practices for Noise-Based Augmentation to Improve the Performance of Deployable Speech-Based Emotion Recognition Systems","date":"2021-06-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-training-of-acoustic-encoders","title":"Collaborative Training of Acoustic Encoders for Speech Recognition","date":"2021-06-16","arxiv_id":"2106.08960","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-classification-on-spoken-documents","title":"Topic Classification on Spoken Documents Using Deep Acoustic and Linguistic Features","date":"2021-06-16","arxiv_id":"2106.08637","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-into-pre-training-strategies-for","title":"A Study into Pre-training Strategies for Spoken Language Understanding on Dysarthric Speech","date":"2021-06-15","arxiv_id":"2106.08313","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-adaptation-for-e-commerce-chatbots-using","title":"ASR Adaptation for E-commerce Chatbots using Cross-Utterance Context and Multi-Task Language Modeling","date":"2021-06-15","arxiv_id":"2106.09532","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialectal-speech-recognition-and-translation","title":"Dialectal Speech Recognition and Translation of Swiss German Speech to Standard German Text: Microsoft's Submission to SwissText 2021","date":"2021-06-15","arxiv_id":"2106.08126","repositories_listed":0,"syntology":null},{"url":null,"slug":"e2e-based-multi-task-learning-approach-to","title":"E2E-based Multi-task Learning Approach to Joint Speech and Accent Recognition","date":"2021-06-15","arxiv_id":"2106.08211","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-opus-compression-for-far-field","title":"Multi-channel Opus compression for far-field automatic speech recognition with a fixed bitrate budget","date":"2021-06-15","arxiv_id":"2106.07994","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-attacks-and-defenses-against-aed","title":"Audio Attacks and Defenses against AED Systems -- A Practical Study","date":"2021-06-14","arxiv_id":"2106.07428","repositories_listed":0,"syntology":null},{"url":null,"slug":"codert-distilling-encoder-representations","title":"CoDERT: Distilling Encoder Representations with Co-learning for Transducer-based Speech Recognition","date":"2021-06-14","arxiv_id":"2106.07734","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-gradient-aggregation-for-federated","title":"Dynamic Gradient Aggregation for Federated Domain Adaptation","date":"2021-06-14","arxiv_id":"2106.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"kaizen-continuously-improving-teacher-using","title":"Kaizen: Continuously improving teacher using Exponential Moving Average for semi-supervised speech recognition","date":"2021-06-14","arxiv_id":"2106.07759","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-domain-mismatch-in-low-resource","title":"Overcoming Domain Mismatch in Low Resource Sequence-to-Sequence ASR Models using Hybrid Generated Pseudotranscripts","date":"2021-06-14","arxiv_id":"2106.07716","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthasr-unlocking-synthetic-data-for-speech","title":"SynthASR: Unlocking Synthetic Data for Speech Recognition","date":"2021-06-14","arxiv_id":"2106.07803","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-heterogeneity-in-semi-supervised","title":"Using heterogeneity in semi-supervised transcription hypotheses to improve code-switched speech recognition","date":"2021-06-14","arxiv_id":"2106.07699","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-sentence-neural-language-models-for","title":"Cross-utterance Reranking Models with BERT and Graph Convolutional Networks for Conversational Speech Recognition","date":"2021-06-13","arxiv_id":"2106.06922","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-t-asr-performance-with-date","title":"Improving RNN-T ASR Performance with Date-Time and Location Awareness","date":"2021-06-11","arxiv_id":"2106.06183","repositories_listed":0,"syntology":null}],"record_sha256":"df6a58186f0f104e91842213e8f8963bd00c70dc598677937bd2b4b10c4593e3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}