{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/22","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":22,"pages_in_order":32,"rows_per_page":100,"rows":[2101,2200],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/21","next":"/task/automatic-speech-recognition-2/papers/23","papers":[{"url":null,"slug":"spell-my-name-keyword-boosted-speech","title":"Spell my name: keyword boosted speech recognition","date":"2021-10-06","arxiv_id":"2110.02791","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-rescoring-and-confidence-estimation-with","title":"ASR Rescoring and Confidence Estimation with ELECTRA","date":"2021-10-05","arxiv_id":"2110.01857","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-contextual-adaptation-with-neural","title":"Fast Contextual Adaptation with Neural Associative Memory for On-Device Personalized Speech Recognition","date":"2021-10-05","arxiv_id":"2110.02220","repositories_listed":0,"syntology":null},{"url":"/paper/is-attention-always-needed-a-case-study-on","slug":"is-attention-always-needed-a-case-study-on","title":"Is Attention always needed? A Case Study on Language Identification from Speech","date":"2021-10-05","arxiv_id":"2110.03427","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-noisy-audio-dataset-to-evaluate","title":"Building a Noisy Audio Dataset to Evaluate Machine Learning Approaches for Automatic Speech Recognition Systems","date":"2021-10-04","arxiv_id":"2110.01425","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-pre-trained-asr-models-for","title":"Exploiting Pre-Trained ASR Models for Alzheimer's Disease Recognition Through Spontaneous Speech","date":"2021-10-04","arxiv_id":"2110.01493","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-end-to-end-speech","title":"Towards efficient end-to-end speech recognition with biologically-inspired neural networks","date":"2021-10-04","arxiv_id":"2110.02743","repositories_listed":0,"syntology":null},{"url":null,"slug":"chinese-medical-speech-recognition-with","title":"Chinese Medical Speech Recognition with Punctuated Hypothesis","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"employing-low-pass-filtered-temporal-speech","title":"Employing low-pass filtered temporal speech features for the training of ideal ratio mask in speech enhancement","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-integration-of-e2e-asr-and","title":"Exploring the Integration of E2E ASR and Pronunciation Modeling for English Mispronunciation Detection","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-technology-for-everyone-automatic","title":"Speech Technology for Everyone: Automatic Speech Recognition for Non-Native English with Transfer Learning","date":"2021-10-01","arxiv_id":"2110.00678","repositories_listed":0,"syntology":null},{"url":null,"slug":"spliceout-a-simple-and-efficient-audio","title":"SpliceOut: A Simple and Efficient Audio Augmentation Method","date":"2021-09-30","arxiv_id":"2110.00046","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-self-supervised-speech-pre","title":"Comparison of Self-Supervised Speech Pre-Training Methods on Flemish Dutch","date":"2021-09-29","arxiv_id":"2109.14357","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-sequence-to-sequence-networks","title":"Conditioning Sequence-to-sequence Networks with Learned Activations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-limited-adversarial","title":"Demystifying Limited Adversarial Transferability in Automatic Speech Recognition Systems","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mlp-based-architecture-with-variable-length","title":"MLP-based architecture with variable length input for automatic speech recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"phasefool-phase-oriented-audio-adversarial","title":"PhaseFool: Phase-oriented Audio Adversarial Examples via Energy Dissipation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesising-audio-adversarial-examples-for","title":"Synthesising Audio Adversarial Examples for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-role-of-self-attention-for","title":"Understanding the Role of Self Attention for Efficient Speech Recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"w-ctc-a-connectionist-temporal-classification","title":"W-CTC: a Connectionist Temporal Classification Loss with Wild Cards","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"private-language-model-adaptation-for-speech","title":"Private Language Model Adaptation for Speech Recognition","date":"2021-09-28","arxiv_id":"2110.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-level-confidence-estimation-for-rnn","title":"Word-level confidence estimation for RNN transducers","date":"2021-09-28","arxiv_id":"2110.15222","repositories_listed":0,"syntology":null},{"url":"/paper/bigssl-exploring-the-frontier-of-large-scale","slug":"bigssl-exploring-the-frontier-of-large-scale","title":"BigSSL: Exploring the Frontier of Large-Scale Semi-Supervised Learning for Automatic Speech Recognition","date":"2021-09-27","arxiv_id":"2109.13226","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-and-opportunities-of-speech","title":"Challenges and Opportunities of Speech Recognition for Bengali Language","date":"2021-09-27","arxiv_id":"2109.13217","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-model-robustness-to-automatic-speech","title":"Topic Model Robustness to Automatic Speech Recognition Errors in Podcast Transcripts","date":"2021-09-25","arxiv_id":"2109.12306","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-domain-specific-language-models-for","title":"Learning Domain Specific Language Models for Automatic Speech Recognition through Machine Translation","date":"2021-09-21","arxiv_id":"2110.10261","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-recognition-is-worth-32","title":"Audio-Visual Speech Recognition is Worth 32$\\times$32$\\times$8 Voxels","date":"2021-09-20","arxiv_id":"2109.09536","repositories_listed":0,"syntology":null},{"url":null,"slug":"irnn-integer-only-recurrent-neural-network","title":"iRNN: Integer-only Recurrent Neural Network","date":"2021-09-20","arxiv_id":"2109.09828","repositories_listed":0,"syntology":null},{"url":null,"slug":"meetdot-videoconferencing-with-live","title":"MeetDot: Videoconferencing with Live Translation Captions","date":"2021-09-20","arxiv_id":"2109.09577","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-approach-for-measuring-the","title":"Model-Based Approach for Measuring the Fairness in ASR","date":"2021-09-19","arxiv_id":"2109.09061","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-audio-textual-architecture-for","title":"Multimodal Audio-textual Architecture for Robust Spoken Language Understanding","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pdaugment-data-augmentation-by-pitch-and","title":"PDAugment: Data Augmentation by Pitch and Duration Adjustments for Automatic Lyrics Transcription","date":"2021-09-16","arxiv_id":"2109.07940","repositories_listed":0,"syntology":null},{"url":null,"slug":"utterance-level-neural-confidence-measure-for","title":"Utterance-level neural confidence measure for end-to-end children speech recognition","date":"2021-09-16","arxiv_id":"2109.07750","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accent-identification-and-accented","title":"Improving Accent Identification and Accented Speech Recognition Under a Framework of Self-supervised Learning","date":"2021-09-15","arxiv_id":"2109.07349","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-streaming-transformer-based-asr","title":"Improving Streaming Transformer Based ASR Under a Framework of Self-supervised Learning","date":"2021-09-15","arxiv_id":"2109.07327","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-transformer-with-unified","title":"Non-autoregressive Transformer with Unified Bidirectional Decoder for Automatic Speech Recognition","date":"2021-09-14","arxiv_id":"2109.06684","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-adapters-for-parameter-efficient-asr","title":"Residual Adapters for Parameter-Efficient ASR Adaptation to Atypical and Accented Speech","date":"2021-09-14","arxiv_id":"2109.06952","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-schemes-for","title":"Unsupervised Domain Adaptation Schemes for Building ASR in Low-resource Languages","date":"2021-09-12","arxiv_id":"2109.05494","repositories_listed":0,"syntology":null},{"url":null,"slug":"remember-the-context-asr-slot-error","title":"Remember the context! ASR slot error correction through memorization","date":"2021-09-10","arxiv_id":"2109.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-attention-channel-combinator-frontend","title":"Self-Attention Channel Combinator Frontend for End-to-End Multichannel Far-field Speech Recognition","date":"2021-09-10","arxiv_id":"2109.04783","repositories_listed":0,"syntology":null},{"url":null,"slug":"coarse-to-fine-and-cross-lingual-asr-transfer","title":"Coarse-To-Fine And Cross-Lingual ASR Transfer","date":"2021-09-02","arxiv_id":"2109.00916","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-end-to-end-automatic-speech-1","title":"Robustness of end-to-end Automatic Speech Recognition Models – A Case Study using Mozilla DeepSpeech","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-constrained-pointer-generator-for-end-to","title":"Tree-constrained Pointer Generator for End-to-end Contextual Speech Recognition","date":"2021-09-01","arxiv_id":"2109.00627","repositories_listed":0,"syntology":null},{"url":"/paper/asr-glue-a-new-multi-task-benchmark-for-asr","slug":"asr-glue-a-new-multi-task-benchmark-for-asr","title":"ASR-GLUE: A New Multi-task Benchmark for ASR-Robust Natural Language Understanding","date":"2021-08-30","arxiv_id":"2108.13048","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigations-on-speech-recognition-systems","title":"Investigations on Speech Recognition Systems for Low-Resource Dialectal Arabic-English Code-Switching Speech","date":"2021-08-29","arxiv_id":"2108.12881","repositories_listed":0,"syntology":null},{"url":null,"slug":"4-bit-quantization-of-lstm-based-speech","title":"4-bit Quantization of LSTM-based Speech Recognition Models","date":"2021-08-27","arxiv_id":"2108.12074","repositories_listed":0,"syntology":null},{"url":null,"slug":"grammar-based-identification-of-speaker-role","title":"Grammar Based Speaker Role Identification for Air Traffic Control Speech Recognition","date":"2021-08-27","arxiv_id":"2108.12175","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-callsign-recognition-with-air","title":"Improving callsign recognition with air-surveillance data in air-traffic communication","date":"2021-08-27","arxiv_id":"2108.12156","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-aware-warping-factors-in-mask-based","title":"Task-aware Warping Factors in Mask-based Speech Enhancement","date":"2021-08-27","arxiv_id":"2108.12128","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-single-channel-speech","title":"Cross-domain Single-channel Speech Enhancement Model with Bi-projection Fusion Module for Noise-robust ASR","date":"2021-08-26","arxiv_id":"2108.11598","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-exposure-bias-in-training-recurrent","title":"Reducing Exposure Bias in Training Recurrent Neural Network Transducers","date":"2021-08-24","arxiv_id":"2108.10803","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-transformer-based-framework-for","title":"A Unified Transformer-based Framework for Duplex Text Normalization","date":"2021-08-23","arxiv_id":"2108.09889","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-using-limited","title":"Automatic Speech Recognition And Limited Vocabulary: A Survey","date":"2021-08-23","arxiv_id":"2108.10254","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-summarization-for-longform","title":"Hierarchical Summarization for Longform Spoken Dialog","date":"2021-08-21","arxiv_id":"2108.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-streams-and-two-resolution-spectrograms","title":"A Multi-level Acoustic Feature Extraction Framework for Transformer Based End-to-End Speech Recognition","date":"2021-08-18","arxiv_id":"2108.07980","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-light-weight-contextual-spelling-correction","title":"A Light-weight contextual spelling correction model for customizing transducer-based speech recognition systems","date":"2021-08-17","arxiv_id":"2108.07493","repositories_listed":0,"syntology":null},{"url":null,"slug":"stargan-vc-asr-stargan-based-non-parallel","title":"StarGAN-VC+ASR: StarGAN-based Non-Parallel Voice Conversion Regularized by Automatic Speech Recognition","date":"2021-08-10","arxiv_id":"2108.04395","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hw-tsc-s-offline-speech-translation","title":"The HW-TSC's Offline Speech Translation Systems for IWSLT 2021 Evaluation","date":"2021-08-09","arxiv_id":"2108.03845","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-and-neural-network-guided-convolutional","title":"Blind and neural network-guided convolutional beamformer for joint denoising, dereverberation, and source separation","date":"2021-08-04","arxiv_id":"2108.01836","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyn-asr-compact-multilingual-speech","title":"Dyn-ASR: Compact, Multilingual Speech Recognition via Spoken Language and Accent Identification","date":"2021-08-04","arxiv_id":"2108.02034","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-distinction-between-asr-errors-and","title":"Improving Distinction between ASR Errors and Speech Disfluencies with Feature Space Interpolation","date":"2021-08-04","arxiv_id":"2108.01812","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-in-speech","title":"Unsupervised Domain Adaptation in Speech Recognition using Phonetic Features","date":"2021-08-04","arxiv_id":"2108.02850","repositories_listed":0,"syntology":null},{"url":"/paper/amortized-neural-networks-for-low-latency","slug":"amortized-neural-networks-for-low-latency","title":"Amortized Neural Networks for Low-Latency Speech Recognition","date":"2021-08-03","arxiv_id":"2108.01553","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-neural-diff-for-speech-models","title":"Learning a Neural Diff for Speech Models","date":"2021-08-03","arxiv_id":"2108.01561","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-recognition-of-suprasegmentals-in","title":"Automatic recognition of suprasegmentals in speech","date":"2021-08-02","arxiv_id":"2108.01122","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupling-recognition-and-transcription-in","title":"Decoupling recognition and transcription in Mandarin ASR","date":"2021-08-02","arxiv_id":"2108.01129","repositories_listed":0,"syntology":null},{"url":null,"slug":"bts-back-transcription-for-speech-to-text","title":"BTS: Back TranScription for Speech-to-Text Post-Processor using Text-to-Speech-to-Text","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-might-we-create-better-benchmarks-for","title":"How Might We Create Better Benchmarks for Speech Recognition?","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ims-systems-for-the-iwslt-2021-low-resource-1","title":"IMS’ Systems for the IWSLT 2021 Low-Resource Speech Translation Task","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-reinforcement-learning-for-table","title":"Interactive Reinforcement Learning for Table Balancing Robot","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-knowledge-distillation-for-translating","title":"On Knowledge Distillation for Translating Erroneous Speech Transcriptions","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-trac-systems-for-the-iwslt-2021-low","title":"ON-TRAC’ systems for the IWSLT 2021 low-resource speech translation and multilingual speech translation shared tasks","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large-1","title":"QASR: QCRI Aljazeera Speech Resource A Large Scale Annotated Arabic Speech Corpus","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"technology-augmented-multilingual","title":"Technology-Augmented Multilingual Communication Models: New Interaction Paradigms, Shifts in the Language Services Industry, and Implications for Training Programs","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"without-further-ado-direct-and-simultaneous","title":"Without Further Ado: Direct and Simultaneous Speech Translation by AppTek in 2021","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zjus-iwslt-2021-speech-translation-system","title":"ZJU’s IWSLT 2021 Speech Translation System","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"can-you-hear-it-backdoor-attacks-via","title":"Can You Hear It? Backdoor Attacks via Ultrasonic Triggers","date":"2021-07-30","arxiv_id":"2107.14569","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-gpt-gpt-2-and-bert-language-models","title":"Adapting GPT, GPT-2 and BERT Language Models for Speech Recognition","date":"2021-07-29","arxiv_id":"2108.07789","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-wav2vec2-an-application-of","title":"An Adapter Based Pre-Training for Efficient and Scalable Self-Supervised Speech Representation Learning","date":"2021-07-26","arxiv_id":"2107.13530","repositories_listed":0,"syntology":null},{"url":null,"slug":"facetron-multi-speaker-face-to-speech-model","title":"Facetron: A Multi-speaker Face-to-Speech Model based on Cross-modal Latent Representations","date":"2021-07-26","arxiv_id":"2107.12003","repositories_listed":0,"syntology":null},{"url":"/paper/olr-2021-challenge-datasets-rules-and","slug":"olr-2021-challenge-datasets-rules-and","title":"OLR 2021 Challenge: Datasets, Rules and Baselines","date":"2021-07-23","arxiv_id":"2107.11113","repositories_listed":0,"syntology":null},{"url":null,"slug":"carnelinet-neural-mixture-model-for-automatic","title":"CarneliNet: Neural Mixture Model for Automatic Speech Recognition","date":"2021-07-22","arxiv_id":"2107.10708","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitask-based-joint-learning-approach-to","title":"Multitask-Based Joint Learning Approach To Robust ASR For Radio Communication Speech","date":"2021-07-22","arxiv_id":"2107.10701","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-prosody-modeling-for-asr-tts-based-voice","title":"On Prosody Modeling for ASR+TTS based Voice Conversion","date":"2021-07-20","arxiv_id":"2107.09477","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-baseline-model-for-computationally","title":"A baseline model for computationally inexpensive speech recognition for Kazakh using the Coqui STT framework","date":"2021-07-19","arxiv_id":"2107.10637","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-with-cross-attention-for","title":"Multi-task Learning with Cross Attention for Keyword Spotting","date":"2021-07-15","arxiv_id":"2107.07634","repositories_listed":0,"syntology":null},{"url":null,"slug":"vad-free-streaming-hybrid-ctc-attention-asr","title":"VAD-free Streaming Hybrid CTC/Attention ASR for Unsegmented Recording","date":"2021-07-15","arxiv_id":"2107.07509","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-configurable-multilingual-model-is-all-you","title":"A Configurable Multilingual Model is All You Need to Recognize All Languages","date":"2021-07-13","arxiv_id":"2107.05876","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-iwslt-2021-but-speech-translation-systems","title":"The IWSLT 2021 BUT Speech Translation Systems","date":"2021-07-13","arxiv_id":"2107.06155","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-speech-translation","title":"Zero-shot Speech Translation","date":"2021-07-13","arxiv_id":"2107.06010","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptual-based-deep-learning-denoiser-as-a","title":"Perceptual-based deep-learning denoiser as a defense against adversarial attacks on ASR systems","date":"2021-07-12","arxiv_id":"2107.05222","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-prediction-end-to-end-active-learning","title":"Loss Prediction: End-to-End Active Learning Approach For Speech Recognition","date":"2021-07-09","arxiv_id":"2107.04289","repositories_listed":0,"syntology":null},{"url":null,"slug":"noisy-training-improves-e2e-asr-for-the-edge","title":"Noisy Training Improves E2E ASR for the Edge","date":"2021-07-09","arxiv_id":"2107.04677","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-lattice-free-boosted-mmi-training-of-hmm","title":"On lattice-free boosted MMI training of HMM and CTC-based full-context ASR models","date":"2021-07-09","arxiv_id":"2107.04154","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-language-identification-through","title":"Improved Language Identification Through Cross-Lingual Self-Supervised Learning","date":"2021-07-08","arxiv_id":"2107.04082","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-rich-transcription-style-automatic","title":"End-to-End Rich Transcription-Style Automatic Speech Recognition with Semi-Supervised Learning","date":"2021-07-07","arxiv_id":"2107.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-modular-and-joint","title":"A Comparative Study of Modular and Joint Approaches for Speaker-Attributed ASR on Monaural Long-Form Audio","date":"2021-07-06","arxiv_id":"2107.02852","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-practical-aspects-of-single","title":"Investigation of Practical Aspects of Single Channel Speech Separation for ASR","date":"2021-07-05","arxiv_id":"2107.01922","repositories_listed":0,"syntology":null},{"url":null,"slug":"arabic-code-switching-speech-recognition","title":"Arabic Code-Switching Speech Recognition using Monolingual Data","date":"2021-07-04","arxiv_id":"2107.01573","repositories_listed":0,"syntology":null}],"record_sha256":"26915e32da097eb44caeaa9478d8bb38cac957ff6de74d1badbed8e812095579","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}