{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/33","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":33,"pages_in_order":65,"rows_per_page":100,"rows":[3201,3300],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/32","next":"/task/speech-recognition/papers/34","papers":[{"url":null,"slug":"online-model-compression-for-federated","title":"Online Model Compression for Federated Learning with Large Models","date":"2022-05-06","arxiv_id":"2205.03494","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-a-novel-korean-learning-application","title":"Design of a novel Korean learning application for efficient pronunciation correction","date":"2022-05-04","arxiv_id":"2205.02001","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-trac-consortium-systems-for-the-iwslt-2022","title":"ON-TRAC Consortium Systems for the IWSLT 2022 Dialect and Low-resource Speech Translation Tasks","date":"2022-05-04","arxiv_id":"2205.01987","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-monoaural-speech-enhancement-for-automatic","title":"On monoaural speech enhancement for automatic recognition of real noisy speech using mixture invariant training","date":"2022-05-03","arxiv_id":"2205.01751","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-meeting-transcription-system-for-an-ad-hoc","title":"A Meeting Transcription System for an Ad-Hoc Acoustic Sensor Network","date":"2022-05-02","arxiv_id":"2205.00944","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-speech-driven-lip-sync-model-with-cnn","title":"A Novel Speech-Driven Lip-Sync Model with CNN and LSTM","date":"2022-05-02","arxiv_id":"2205.00916","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilingual-end-to-end-asr-with-byte-level","title":"Bilingual End-to-End ASR with Byte-Level Subwords","date":"2022-05-01","arxiv_id":"2205.00485","repositories_listed":0,"syntology":null},{"url":null,"slug":"cmus-iwslt-2022-dialect-speech-translation","title":"CMU’s IWSLT 2022 Dialect Speech Translation System","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"corpus-development-of-kiswahili-speech","title":"Corpus Development of Kiswahili Speech Recognition Test and Evaluation sets, Preemptively Mitigating Demographic Bias Through Collaboration with Linguists","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"discourse-on-asr-measurement-introducing-the","title":"Discourse on ASR Measurement: Introducing the ARPOCA Assessment Tool","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-documentation-of-hupa-with","title":"Enhancing Documentation of Hupa with Automatic Speech Recognition","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-shared-task-on-speech","title":"Findings of the Shared Task on Speech Recognition for Vulnerable Individuals in Tamil","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-pre-trained-models-for-automatic","title":"Fine-tuning pre-trained models for Automatic Speech Recognition, experiments on a fieldwork corpus of Japhug (Trans-Himalayan family)","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jhu-iwslt-2022-dialect-speech-translation","title":"JHU IWSLT 2022 Dialect Speech Translation System Description","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mtl-slt-multi-task-learning-for-spoken","title":"MTL-SLT: Multi-Task Learning for Spoken Language Tasks","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-fusion-via-cortical-network","title":"Multimodal fusion via cortical network inspired losses","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nvidia-nemo-offline-speech-translation","title":"NVIDIA NeMo Offline Speech Translation Systems for IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"phoneme-transcription-of-endangered-languages","title":"Phoneme transcription of endangered languages: an evaluation of recent ASR architectures in the single speaker scenario","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-semantic-driven-phoneme-1","title":"Self-supervised Semantic-driven Phoneme Discovery for Zero-resource Speech Recognition","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ssncse-nlp-lt-edi-acl2022-speech-recognition","title":"SSNCSE_NLP@LT-EDI-ACL2022: Speech Recognition for Vulnerable Individuals in Tamil using pre-trained XLSR models","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"suh-asr-lt-edi-acl2022-transformer-based","title":"SUH_ASR@LT-EDI-ACL2022: Transformer based Approach for Speech Recognition for Vulnerable Individuals in Tamil","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hw-tscs-offline-speech-translation-system","title":"The HW-TSC’s Offline Speech Translation System for IWSLT 2022 Evaluation","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-xiaomi-text-to-text-simultaneous-speech","title":"The Xiaomi Text-to-Text Simultaneous Speech Translation System for IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"why-does-self-supervised-learning-for-speech","title":"Why does Self-Supervised Learning for Speech Recognition Benefit Speaker Recognition?","date":"2022-04-27","arxiv_id":"2204.12765","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-scalar-prediction-for-improving-robust","title":"Mask scalar prediction for improving robust automatic speech recognition","date":"2022-04-26","arxiv_id":"2204.12092","repositories_listed":0,"syntology":null},{"url":null,"slug":"cleanformer-a-microphone-array-configuration","title":"Cleanformer: A multichannel array configuration-invariant neural enhancement frontend for ASR in smart speakers","date":"2022-04-25","arxiv_id":"2204.11933","repositories_listed":0,"syntology":null},{"url":null,"slug":"enable-deep-learning-on-mobile-devices","title":"Enable Deep Learning on Mobile Devices: Methods, Systems, and Applications","date":"2022-04-25","arxiv_id":"2204.11786","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-attention-in-sequence-to-sequence","title":"Supervised Attention in Sequence-to-Sequence Models for Speech Recognition","date":"2022-04-25","arxiv_id":"2204.12308","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-far-field-speech-recognition-using","title":"Improved far-field speech recognition using Joint Variational Autoencoder","date":"2022-04-24","arxiv_id":"2204.11286","repositories_listed":0,"syntology":null},{"url":null,"slug":"e2e-segmenter-joint-segmenting-and-decoding","title":"E2E Segmenter: Joint Segmenting and Decoding for Long-Form ASR","date":"2022-04-22","arxiv_id":"2204.10749","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-training-of-neural-transducer-for","title":"Efficient Training of Neural Transducer for Speech Recognition","date":"2022-04-22","arxiv_id":"2204.10586","repositories_listed":0,"syntology":null},{"url":null,"slug":"wabert-a-low-resource-end-to-end-model-for","title":"WaBERT: A Low-resource End-to-end Model for Spoken Language Understanding and Speech-to-BERT Alignment","date":"2022-04-22","arxiv_id":"2204.10461","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-testing-of-data-and-knowledge","title":"Robustness Testing of Data and Knowledge Driven Anomaly Detection in Cyber-Physical Systems","date":"2022-04-20","arxiv_id":"2204.09183","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-monotonic-transducers-for","title":"An Investigation of Monotonic Transducers for Large-Scale Automatic Speech Recognition","date":"2022-04-19","arxiv_id":"2204.08858","repositories_listed":0,"syntology":null},{"url":null,"slug":"blockwise-streaming-transformer-for-spoken","title":"Blockwise Streaming Transformer for Spoken Language Understanding and Simultaneous Speech Translation","date":"2022-04-19","arxiv_id":"2204.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"disappeared-command-spoofing-attack-on","title":"Disappeared Command: Spoofing Attack On Automatic Speech Recognition Systems with Sound Masking","date":"2022-04-19","arxiv_id":"2204.08977","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-speech-tools-for-helping","title":"Automated speech tools for helping communities process restricted-access corpora for language revival efforts","date":"2022-04-15","arxiv_id":"2204.07272","repositories_listed":0,"syntology":null},{"url":null,"slug":"lombard-effect-for-bilingual-speakers-in","title":"Lombard Effect for Bilingual Speakers in Cantonese and English: importance of spectro-temporal features","date":"2022-04-14","arxiv_id":"2204.06907","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-cascaded-encoder-asr-model-for","title":"A Unified Cascaded Encoder ASR Model for Dynamic Model Sizes","date":"2022-04-13","arxiv_id":"2204.06164","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-critical-sequence-training-for-automatic","title":"Self-critical Sequence Training for Automatic Speech Recognition","date":"2022-04-13","arxiv_id":"2204.06260","repositories_listed":0,"syntology":null},{"url":null,"slug":"study-of-indian-english-pronunciation","title":"Study of Indian English Pronunciation Variabilities relative to Received Pronunciation","date":"2022-04-13","arxiv_id":"2204.06502","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-in-german-a-detailed-error-analysis","title":"ASR in German: A Detailed Error Analysis","date":"2022-04-12","arxiv_id":"2204.05617","repositories_listed":0,"syntology":null},{"url":null,"slug":"correctspeech-a-fully-automated-system-for","title":"CorrectSpeech: A Fully Automated System for Speech Correction and Accent Reduction","date":"2022-04-12","arxiv_id":"2204.05460","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-an-asr-error-robust-spoken-virtual","title":"Building an ASR Error Robust Spoken Virtual Patient System in a Highly Class-Imbalanced Scenario Without Speech Data","date":"2022-04-11","arxiv_id":"2204.05183","repositories_listed":0,"syntology":null},{"url":null,"slug":"multistream-neural-architectures-for-cued","title":"Multistream neural architectures for cued-speech recognition using a pre-trained visual feature extractor and constrained CTC decoding","date":"2022-04-11","arxiv_id":"2204.04965","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-speech-text-pre-training-for-speech-1","title":"Unified Speech-Text Pre-training for Speech Translation and Recognition","date":"2022-04-11","arxiv_id":"2204.05409","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-embeddings-for-robust-user-based-amateur","title":"Deep Embeddings for Robust User-Based Amateur Vocal Percussion Classification","date":"2022-04-10","arxiv_id":"2204.04646","repositories_listed":0,"syntology":null},{"url":null,"slug":"adding-connectionist-temporal-summarization","title":"Adding Connectionist Temporal Summarization into Conformer to Improve Its Decoder Efficiency For Speech Recognition","date":"2022-04-08","arxiv_id":"2204.03889","repositories_listed":0,"syntology":null},{"url":null,"slug":"auditory-based-data-augmentation-for-end-to","title":"Auditory-Based Data Augmentation for End-to-End Automatic Speech Recognition","date":"2022-04-08","arxiv_id":"2204.04284","repositories_listed":0,"syntology":null},{"url":null,"slug":"defense-against-adversarial-attacks-on-hybrid","title":"Defense against Adversarial Attacks on Hybrid Speech Recognition using Joint Adversarial Fine-tuning with Denoiser","date":"2022-04-08","arxiv_id":"2204.03851","repositories_listed":0,"syntology":null},{"url":null,"slug":"personal-vad-2-0-optimizing-personal-voice","title":"Personal VAD 2.0: Optimizing Personal Voice Activity Detection for On-Device Speech Recognition","date":"2022-04-08","arxiv_id":"2204.03793","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-dysfluencies-in-stuttering-therapy","title":"Detecting Dysfluencies in Stuttering Therapy Using wav2vec 2.0","date":"2022-04-07","arxiv_id":"2204.03417","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-deep-learning-for-all-in-edge","title":"Enabling All In-Edge Deep Learning: A Literature Review","date":"2022-04-07","arxiv_id":"2204.03326","repositories_listed":0,"syntology":null},{"url":null,"slug":"maestro-matched-speech-text-representations","title":"MAESTRO: Matched Speech Text Representations through Modality Matching","date":"2022-04-07","arxiv_id":"2204.03409","repositories_listed":0,"syntology":null},{"url":null,"slug":"three-module-modeling-for-end-to-end-spoken","title":"Three-Module Modeling For End-to-End Spoken Language Understanding Using Pre-trained DNN-HMM-Based Acoustic-Phonetic Model","date":"2022-04-07","arxiv_id":"2204.03315","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-recently-proposed-activation","title":"A survey on recently proposed activation functions for Deep Learning","date":"2022-04-06","arxiv_id":"2204.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-self-supervised-learning-solve-the","title":"A Wav2vec2-Based Experimental Study on Self-Supervised Learning Methods to Improve Child Speech Recognition","date":"2022-04-06","arxiv_id":"2204.05419","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-direct-speech-to-speech-translation","title":"Enhanced Direct Speech-to-Speech Translation Using Self-supervised Pre-training and Data Augmentation","date":"2022-04-06","arxiv_id":"2204.02967","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-and-effective-unsupervised-speech","title":"Simple and Effective Unsupervised Speech Synthesis","date":"2022-04-06","arxiv_id":"2204.02524","repositories_listed":0,"syntology":null},{"url":null,"slug":"successes-and-critical-failures-of-neural","title":"Successes and critical failures of neural networks in capturing human-like speech recognition","date":"2022-04-06","arxiv_id":"2204.03740","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-complementary-joint-training-approach-using","title":"A Complementary Joint Training Approach Using Unpaired Speech and Text for Low-Resource Automatic Speech Recognition","date":"2022-04-05","arxiv_id":"2204.02023","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-multi-channel-speech-separation","title":"Audio-visual multi-channel speech separation, dereverberation and recognition","date":"2022-04-05","arxiv_id":"2204.01977","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-speech-representation-learning","title":"Disentangled Speech Representation Learning Based on Factorized Hierarchical Variational Autoencoder with Self-Supervised Objective","date":"2022-04-05","arxiv_id":"2204.02166","repositories_listed":0,"syntology":null},{"url":null,"slug":"hear-no-evil-towards-adversarial-robustness","title":"Hear No Evil: Towards Adversarial Robustness of Automatic Speech Recognition via Multi-Task Learning","date":"2022-04-05","arxiv_id":"2204.02381","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-data-selection-via-discrete","title":"Unsupervised Data Selection via Discrete Speech Representation for ASR","date":"2022-04-05","arxiv_id":"2204.01981","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-gender-impact-in-self-supervised","title":"A Study of Gender Impact in Self-supervised Models for Speech-to-Text Systems","date":"2022-04-04","arxiv_id":"2204.01397","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-joint-speech-text-embeddings-for","title":"An Analysis of Semantically-Aligned Speech-Text Embeddings","date":"2022-04-04","arxiv_id":"2204.01235","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-self-supervised-speech","title":"Cross-lingual Self-Supervised Speech Representations for Improved Dysarthric Speech Recognition","date":"2022-04-04","arxiv_id":"2204.01670","repositories_listed":0,"syntology":null},{"url":null,"slug":"deliberation-model-for-on-device-spoken","title":"Deliberation Model for On-Device Spoken Language Understanding","date":"2022-04-04","arxiv_id":"2204.01893","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-speech-representations","title":"Self-Supervised Speech Representations Preserve Speech Characteristics while Anonymizing Voices","date":"2022-04-04","arxiv_id":"2204.01677","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-speech-based-end-to-end-automated-speech","title":"Deep Speech Based End-to-End Automated Speech Recognition (ASR) for Indian-English Accents","date":"2022-04-03","arxiv_id":"2204.00977","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-model-for-named-entity-recognition","title":"End-to-end model for named entity recognition from speech without paired training data","date":"2022-04-02","arxiv_id":"2204.00803","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-real-time-personalized-speech","title":"Fast Real-time Personalized Speech Enhancement: End-to-End Enhancement Network (E3Net) and Knowledge Distillation","date":"2022-04-02","arxiv_id":"2204.00771","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-adaptation-for-wav2vec2-based","title":"Speaker adaptation for Wav2vec2 based dysarthric ASR","date":"2022-04-02","arxiv_id":"2204.00770","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-integration-of-speech-recognition","title":"End-to-End Integration of Speech Recognition, Speech Enhancement, and Self-Supervised Learning Representation","date":"2022-04-01","arxiv_id":"2204.00540","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-multi-speaker-asr-with-independent","title":"End-to-End Multi-speaker ASR with Independent Vector Analysis","date":"2022-04-01","arxiv_id":"2204.00218","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-multi-talker-audio-visual-asr","title":"End-to-end multi-talker audio-visual ASR using an active speaker attention module","date":"2022-04-01","arxiv_id":"2204.00652","repositories_listed":0,"syntology":null},{"url":null,"slug":"filter-based-discriminative-autoencoders-for","title":"Filter-based Discriminative Autoencoders for Children Speech Recognition","date":"2022-04-01","arxiv_id":"2204.00164","repositories_listed":0,"syntology":null},{"url":null,"slug":"interaug-augmenting-noisy-intermediate","title":"InterAug: Augmenting Noisy Intermediate Predictions for CTC-based ASR","date":"2022-04-01","arxiv_id":"2204.00174","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-sequence-intermediate-conditioning-for","title":"Alternate Intermediate Conditioning with Syllable-level and Character-level Targets for Japanese ASR","date":"2022-04-01","arxiv_id":"2204.00175","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-rnn-t-with-semantic-decoder-for","title":"Multi-task RNN-T with Semantic Decoder for Streamable Spoken Language Understanding","date":"2022-04-01","arxiv_id":"2204.00558","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-confidence-gates-for-joint-training","title":"Multiple Confidence Gates For Joint Training Of SE And ASR","date":"2022-04-01","arxiv_id":"2204.00226","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-speech-emotion-recognition","title":"Probing Speech Emotion Recognition Transformers for Linguistic Knowledge","date":"2022-04-01","arxiv_id":"2204.00400","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-data-augmentation-for-low","title":"Text-To-Speech Data Augmentation for Low Resource Speech Recognition","date":"2022-04-01","arxiv_id":"2204.00291","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-cross-lingual-aphasia-detection","title":"Zero-Shot Cross-lingual Aphasia Detection using Automatic Speech Recognition","date":"2022-04-01","arxiv_id":"2204.00448","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-speaker-attributed","title":"A Comparative Study on Speaker-attributed Automatic Speech Recognition in Multi-party Meetings","date":"2022-03-31","arxiv_id":"2203.16834","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-language-model","title":"An Empirical Study of Language Model Integration for Transducer based Speech Recognition","date":"2022-03-31","arxiv_id":"2203.16776","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-factors-affecting-usefulness-of","title":"Analyzing the factors affecting usefulness of Self-Supervised Pre-trained Representations for Speech Recognition","date":"2022-03-31","arxiv_id":"2203.16973","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectiveness-of-text-to-speech-pseudo-labels","title":"Effectiveness of text to speech pseudo labels for forced alignment and cross lingual pretrained models for low resource speech recognition","date":"2022-03-31","arxiv_id":"2203.16823","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-single-channel-speech-for-multi-1","title":"Exploiting Single-Channel Speech for Multi-Channel End-to-End Speech Recognition: A Comparative Study","date":"2022-03-31","arxiv_id":"2203.16757","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-of-different-temporal-modulations","title":"Importance of Different Temporal Modulations of Speech: A Tale of Two Perspectives","date":"2022-03-31","arxiv_id":"2204.00065","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-language-identification-of-accented","title":"Improving Language Identification of Accented Speech","date":"2022-03-31","arxiv_id":"2203.16972","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-efficient-training-of-rnn-transducer","title":"Memory-Efficient Training of RNN-Transducer with Sampled Softmax","date":"2022-03-31","arxiv_id":"2203.16868","repositories_listed":0,"syntology":null},{"url":"/paper/open-source-magicdata-ramc-a-rich-annotated","slug":"open-source-magicdata-ramc-a-rich-annotated","title":"Open Source MagicData-RAMC: A Rich Annotated Mandarin Conversational(RAMC) Speech Dataset","date":"2022-03-31","arxiv_id":"2203.16844","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switched-and-code-mixed-speech","title":"Code Switched and Code Mixed Speech Recognition for Indic languages","date":"2022-03-30","arxiv_id":"2203.16578","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-domain-adaptation-for-asr-with-full","title":"Federated Domain Adaptation for ASR with Full Self-Supervision","date":"2022-03-30","arxiv_id":"2203.15966","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-for-indic","title":"Improving Speech Recognition for Indic Languages using Language Model","date":"2022-03-30","arxiv_id":"2203.16595","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-word-error-rate-a-good-evaluation-metric","title":"Is Word Error Rate a good evaluation metric for Speech Recognition in Indic Languages?","date":"2022-03-30","arxiv_id":"2203.16601","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-latency-for-ctc-based-streaming","title":"Dynamic Latency for CTC-Based Streaming Automatic Speech Recognition With Emformer","date":"2022-03-29","arxiv_id":"2203.15613","repositories_listed":0,"syntology":null},{"url":null,"slug":"frequency-directional-attention-model-for","title":"Frequency-Directional Attention Model for Multilingual Automatic Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15473","repositories_listed":0,"syntology":null}],"record_sha256":"901d2972e6955bec2d7b34623475da3d2a3c1c495acb807502dc433208bb44c1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}