{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/19","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":32,"rows_per_page":100,"rows":[1801,1900],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/18","next":"/task/automatic-speech-recognition-2/papers/20","papers":[{"url":null,"slug":"context-based-out-of-vocabulary-word-recovery","title":"Context-based out-of-vocabulary word recovery for ASR systems in Indian languages","date":"2022-06-09","arxiv_id":"2206.04305","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-dubbing-lip-synchronous-voice-preserving","title":"Face-Dubbing++: Lip-Synchronous, Voice Preserving Translation of Videos","date":"2022-06-09","arxiv_id":"2206.04523","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-encoder-decoder-self-supervised-pre","title":"Joint Encoder-Decoder Self-Supervised Pre-training for ASR","date":"2022-06-09","arxiv_id":"2206.04465","repositories_listed":0,"syntology":null},{"url":null,"slug":"legonn-building-modular-encoder-decoder","title":"LegoNN: Building Modular Encoder-Decoder Models","date":"2022-06-07","arxiv_id":"2206.03318","repositories_listed":0,"syntology":null},{"url":null,"slug":"fednst-federated-noisy-student-training-for","title":"FedNST: Federated Noisy Student Training for Automatic Speech Recognition","date":"2022-06-06","arxiv_id":"2206.02797","repositories_listed":0,"syntology":null},{"url":null,"slug":"pronunciation-dictionary-free-multilingual","title":"Pronunciation Dictionary-Free Multilingual Speech Synthesis by Combining Unsupervised and Supervised Phonetic Representations","date":"2022-06-02","arxiv_id":"2206.00951","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-semi-automated-live-interlingual","title":"A Semi-Automated Live Interlingual Communication Workflow Featuring Intralingual Respeaking: Evaluation and Benchmarking","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-irish-the","title":"Automatic Speech Recognition for Irish: the ABAIR-ÉIST System","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bea-base-a-benchmark-for-asr-of-spontaneous-1","title":"BEA-Base: A Benchmark for ASR of Spontaneous Hungarian","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-open-source-speech-technology-for","title":"Building Open-source Speech Technology for Low-resource Minority Languages with SáMi as an Example – Tools, Methods and Experiments","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-speech-recognition-needs-data","title":"Conversational Speech Recognition Needs Data? Experiments with Austrian German","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-automatic-speech-recognition-for","title":"Developing Automatic Speech Recognition for Scottish Gaelic","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-automatic-speech-recognition","title":"Development of Automatic Speech Recognition for the Documentation of Cook Islands Māori","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-off-the-shelf-speech-1","title":"Evaluation of Off-the-shelf Speech Recognizers on Different Accents in a Dialogue Domain","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-synthetic-clinical-speech-data","title":"Generating Synthetic Clinical Speech Data through Simulated ASR Deletion Error","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"huqariq-a-multilingual-speech-corpus-of-1","title":"Huqariq: A Multilingual Speech Corpus of Native Languages of Peru forSpeech Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"listra-automatic-speech-translation-english-1","title":"LiSTra Automatic Speech Translation: English to Lingala Case Study","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mesures-linguistiques-automatiques-pour","title":"Mesures linguistiques automatiques pour l’évaluation des systèmes de Reconnaissance Automatique de la Parole (Automated linguistic measures for automatic speech recognition systems’ evaluation)","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-transfer-learning-for-children","title":"Multilingual Transfer Learning for Children Automatic Speech Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"parlamentparla-a-speech-corpus-of-catalan","title":"ParlamentParla: A Speech Corpus of Catalan Parliamentary Sessions","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"parlaspeech-hr-a-freely-available-asr-dataset","title":"ParlaSpeech-HR - a Freely Available ASR Dataset for Croatian Bootstrapped from the ParlaMint Corpus","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"post-stroke-speech-transcription-challenge","title":"Post-Stroke Speech Transcription Challenge (Task B): Correctness Detection in Anomia Diagnosis with Imperfect Transcripts","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"progress-in-multilingual-speech-recognition","title":"Progress in Multilingual Speech Recognition for Low Resource Languages Kurmanji Kurdish, Cree and Inuktut","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"samromur-children-an-icelandic-speech-corpus","title":"Samrómur Children: An Icelandic Speech Corpus","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"snow-mountain-dataset-of-audio-recordings-of","title":"Snow Mountain: Dataset of Audio Recordings of The Bible in Low Resource Languages","date":"2022-06-01","arxiv_id":"2206.01205","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-asr-system-for-the-armenian","title":"Towards a Unified ASR System for the Armenian Standards","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-open-source-dutch-speech","title":"Towards an Open-Source Dutch Speech Recognition System for the Healthcare Domain","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-synthesis-based-data-augmentation","title":"Adversarial synthesis based data-augmentation for code-switched spoken language identification","date":"2022-05-30","arxiv_id":"2205.15747","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-activation-network-for-low-resource","title":"Adaptive Activation Network For Low Resource Multilingual Speech Recognition","date":"2022-05-28","arxiv_id":"2205.14326","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-to-articulatory-speech-inversion-1","title":"Acoustic-to-articulatory Speech Inversion with Multi-task Learning","date":"2022-05-27","arxiv_id":"2205.13755","repositories_listed":0,"syntology":null},{"url":null,"slug":"punctuation-restoration-in-spanish-customer","title":"Punctuation Restoration in Spanish Customer Support Transcripts using Transfer Learning","date":"2022-05-27","arxiv_id":"2205.13961","repositories_listed":0,"syntology":null},{"url":null,"slug":"clinical-dialogue-transcription-error","title":"Clinical Dialogue Transcription Error Correction using Seq2Seq Models","date":"2022-05-26","arxiv_id":"2205.13572","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-adapters-for-personalized-speech","title":"Contextual Adapters for Personalized Speech Recognition in Neural Transducers","date":"2022-05-26","arxiv_id":"2205.13660","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-training-of-speech-enhancement-and-self","title":"Joint Training of Speech Enhancement and Self-supervised Model for Noise-robust ASR","date":"2022-05-26","arxiv_id":"2205.13293","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-on-applying-acoustic-feature","title":"An Investigation on Applying Acoustic Feature Conversion to ASR of Adult and Child Speech","date":"2022-05-25","arxiv_id":"2205.12477","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-reservoir-computing-models-for","title":"Heterogeneous Reservoir Computing Models for Persian Speech Recognition","date":"2022-05-25","arxiv_id":"2205.12594","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-ctc-based-asr-models-with-gated","title":"Improving CTC-based ASR Models with Gated Interlayer Collaboration","date":"2022-05-25","arxiv_id":"2205.12462","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-lexical-replacements-for-arabic","title":"Investigating Lexical Replacements for Arabic-English Code-Switched Data Augmentation","date":"2022-05-25","arxiv_id":"2205.12649","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-building-spoken-language-understanding","title":"On Building Spoken Language Understanding Systems for Low Resourced Languages","date":"2022-05-25","arxiv_id":"2205.12818","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-modeling-units-for-end-to-end","title":"Multi-Level Modeling Units for End-to-End Mandarin Speech Recognition","date":"2022-05-24","arxiv_id":"2205.11998","repositories_listed":0,"syntology":null},{"url":null,"slug":"calibrate-and-refine-a-novel-and-agile","title":"Calibrate and Refine! A Novel and Agile Framework for ASR-error Robust Intent Detection","date":"2022-05-23","arxiv_id":"2205.11008","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-speech-representation","title":"Self-Supervised Speech Representation Learning: A Review","date":"2022-05-21","arxiv_id":"2205.10643","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-spoken-language-identification-1","title":"Automatic Spoken Language Identification using a Time-Delay Neural Network","date":"2022-05-19","arxiv_id":"2205.09564","repositories_listed":0,"syntology":null},{"url":null,"slug":"insights-on-neural-representations-for-end-to","title":"Insights on Neural Representations for End-to-End Speech Recognition","date":"2022-05-19","arxiv_id":"2205.09456","repositories_listed":0,"syntology":null},{"url":null,"slug":"deploying-self-supervised-learning-in-the","title":"Deploying self-supervised learning in the wild for hybrid automatic speech recognition","date":"2022-05-17","arxiv_id":"2205.08598","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-noise-context-aware-enhancement-for","title":"Streaming Noise Context Aware Enhancement For Automatic Speech Recognition in Multi-Talker Environments","date":"2022-05-17","arxiv_id":"2205.08555","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-consistency-training-for-semi","title":"Improved Consistency Training for Semi-Supervised Sequence-to-Sequence ASR via Speech Chain Reconstruction and Self-Transcribing","date":"2022-05-14","arxiv_id":"2205.06963","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretraining-approaches-for-spoken-language","title":"Pretraining Approaches for Spoken Language Recognition: TalTech Submission to the OLR 2021 Challenge","date":"2022-05-14","arxiv_id":"2205.07083","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-adversarial-data-augmentation","title":"Personalized Adversarial Data Augmentation for Dysarthric and Elderly Speech Recognition","date":"2022-05-13","arxiv_id":"2205.06445","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-modeling-of-multi-domain-multi-device","title":"Unified Modeling of Multi-Domain Multi-Device ASR Systems","date":"2022-05-13","arxiv_id":"2205.06655","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-audio-visual-multi-person","title":"A Closer Look at Audio-Visual Multi-Person Speech Recognition and Active Speaker Selection","date":"2022-05-11","arxiv_id":"2205.05684","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-multi-person-audio-visual","title":"End-to-End Multi-Person Audio/Visual Automatic Speech Recognition","date":"2022-05-11","arxiv_id":"2205.05586","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-of-both-worlds-multi-task-audio-visual","title":"Best of Both Worlds: Multi-task Audio-Visual Automatic Speech Recognition and Active Speaker Detection","date":"2022-05-10","arxiv_id":"2205.05206","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-reinforcement-using-target-source","title":"Speaker Reinforcement Using Target Source Extraction for Robust Automatic Speech Recognition","date":"2022-05-09","arxiv_id":"2205.04433","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-conformer-based-waveform-domain-neural","title":"A Conformer-based Waveform-domain Neural Acoustic Echo Canceller Optimized for ASR Accuracy","date":"2022-05-06","arxiv_id":"2205.03481","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-trac-consortium-systems-for-the-iwslt-2022","title":"ON-TRAC Consortium Systems for the IWSLT 2022 Dialect and Low-resource Speech Translation Tasks","date":"2022-05-04","arxiv_id":"2205.01987","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-meeting-transcription-system-for-an-ad-hoc","title":"A Meeting Transcription System for an Ad-Hoc Acoustic Sensor Network","date":"2022-05-02","arxiv_id":"2205.00944","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilingual-end-to-end-asr-with-byte-level","title":"Bilingual End-to-End ASR with Byte-Level Subwords","date":"2022-05-01","arxiv_id":"2205.00485","repositories_listed":0,"syntology":null},{"url":null,"slug":"discourse-on-asr-measurement-introducing-the","title":"Discourse on ASR Measurement: Introducing the ARPOCA Assessment Tool","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-documentation-of-hupa-with","title":"Enhancing Documentation of Hupa with Automatic Speech Recognition","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-shared-task-on-speech","title":"Findings of the Shared Task on Speech Recognition for Vulnerable Individuals in Tamil","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-pre-trained-models-for-automatic","title":"Fine-tuning pre-trained models for Automatic Speech Recognition, experiments on a fieldwork corpus of Japhug (Trans-Himalayan family)","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"jhu-iwslt-2022-dialect-speech-translation","title":"JHU IWSLT 2022 Dialect Speech Translation System Description","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mtl-slt-multi-task-learning-for-spoken","title":"MTL-SLT: Multi-Task Learning for Spoken Language Tasks","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nvidia-nemo-offline-speech-translation","title":"NVIDIA NeMo Offline Speech Translation Systems for IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"phoneme-transcription-of-endangered-languages","title":"Phoneme transcription of endangered languages: an evaluation of recent ASR architectures in the single speaker scenario","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ssncse-nlp-lt-edi-acl2022-speech-recognition","title":"SSNCSE_NLP@LT-EDI-ACL2022: Speech Recognition for Vulnerable Individuals in Tamil using pre-trained XLSR models","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"suh-asr-lt-edi-acl2022-transformer-based","title":"SUH_ASR@LT-EDI-ACL2022: Transformer based Approach for Speech Recognition for Vulnerable Individuals in Tamil","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hw-tscs-offline-speech-translation-system","title":"The HW-TSC’s Offline Speech Translation System for IWSLT 2022 Evaluation","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-xiaomi-text-to-text-simultaneous-speech","title":"The Xiaomi Text-to-Text Simultaneous Speech Translation System for IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-scalar-prediction-for-improving-robust","title":"Mask scalar prediction for improving robust automatic speech recognition","date":"2022-04-26","arxiv_id":"2204.12092","repositories_listed":0,"syntology":null},{"url":null,"slug":"cleanformer-a-microphone-array-configuration","title":"Cleanformer: A multichannel array configuration-invariant neural enhancement frontend for ASR in smart speakers","date":"2022-04-25","arxiv_id":"2204.11933","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-far-field-speech-recognition-using","title":"Improved far-field speech recognition using Joint Variational Autoencoder","date":"2022-04-24","arxiv_id":"2204.11286","repositories_listed":0,"syntology":null},{"url":null,"slug":"wabert-a-low-resource-end-to-end-model-for","title":"WaBERT: A Low-resource End-to-end Model for Spoken Language Understanding and Speech-to-BERT Alignment","date":"2022-04-22","arxiv_id":"2204.10461","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-monotonic-transducers-for","title":"An Investigation of Monotonic Transducers for Large-Scale Automatic Speech Recognition","date":"2022-04-19","arxiv_id":"2204.08858","repositories_listed":0,"syntology":null},{"url":null,"slug":"blockwise-streaming-transformer-for-spoken","title":"Blockwise Streaming Transformer for Spoken Language Understanding and Simultaneous Speech Translation","date":"2022-04-19","arxiv_id":"2204.08920","repositories_listed":0,"syntology":null},{"url":null,"slug":"disappeared-command-spoofing-attack-on","title":"Disappeared Command: Spoofing Attack On Automatic Speech Recognition Systems with Sound Masking","date":"2022-04-19","arxiv_id":"2204.08977","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-speech-tools-for-helping","title":"Automated speech tools for helping communities process restricted-access corpora for language revival efforts","date":"2022-04-15","arxiv_id":"2204.07272","repositories_listed":0,"syntology":null},{"url":null,"slug":"lombard-effect-for-bilingual-speakers-in","title":"Lombard Effect for Bilingual Speakers in Cantonese and English: importance of spectro-temporal features","date":"2022-04-14","arxiv_id":"2204.06907","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-cascaded-encoder-asr-model-for","title":"A Unified Cascaded Encoder ASR Model for Dynamic Model Sizes","date":"2022-04-13","arxiv_id":"2204.06164","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-critical-sequence-training-for-automatic","title":"Self-critical Sequence Training for Automatic Speech Recognition","date":"2022-04-13","arxiv_id":"2204.06260","repositories_listed":0,"syntology":null},{"url":null,"slug":"study-of-indian-english-pronunciation","title":"Study of Indian English Pronunciation Variabilities relative to Received Pronunciation","date":"2022-04-13","arxiv_id":"2204.06502","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-in-german-a-detailed-error-analysis","title":"ASR in German: A Detailed Error Analysis","date":"2022-04-12","arxiv_id":"2204.05617","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-an-asr-error-robust-spoken-virtual","title":"Building an ASR Error Robust Spoken Virtual Patient System in a Highly Class-Imbalanced Scenario Without Speech Data","date":"2022-04-11","arxiv_id":"2204.05183","repositories_listed":0,"syntology":null},{"url":null,"slug":"auditory-based-data-augmentation-for-end-to","title":"Auditory-Based Data Augmentation for End-to-End Automatic Speech Recognition","date":"2022-04-08","arxiv_id":"2204.04284","repositories_listed":0,"syntology":null},{"url":null,"slug":"defense-against-adversarial-attacks-on-hybrid","title":"Defense against Adversarial Attacks on Hybrid Speech Recognition using Joint Adversarial Fine-tuning with Denoiser","date":"2022-04-08","arxiv_id":"2204.03851","repositories_listed":0,"syntology":null},{"url":null,"slug":"three-module-modeling-for-end-to-end-spoken","title":"Three-Module Modeling For End-to-End Spoken Language Understanding Using Pre-trained DNN-HMM-Based Acoustic-Phonetic Model","date":"2022-04-07","arxiv_id":"2204.03315","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-self-supervised-learning-solve-the","title":"A Wav2vec2-Based Experimental Study on Self-Supervised Learning Methods to Improve Child Speech Recognition","date":"2022-04-06","arxiv_id":"2204.05419","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-direct-speech-to-speech-translation","title":"Enhanced Direct Speech-to-Speech Translation Using Self-supervised Pre-training and Data Augmentation","date":"2022-04-06","arxiv_id":"2204.02967","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-complementary-joint-training-approach-using","title":"A Complementary Joint Training Approach Using Unpaired Speech and Text for Low-Resource Automatic Speech Recognition","date":"2022-04-05","arxiv_id":"2204.02023","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-multi-channel-speech-separation","title":"Audio-visual multi-channel speech separation, dereverberation and recognition","date":"2022-04-05","arxiv_id":"2204.01977","repositories_listed":0,"syntology":null},{"url":null,"slug":"hear-no-evil-towards-adversarial-robustness","title":"Hear No Evil: Towards Adversarial Robustness of Automatic Speech Recognition via Multi-Task Learning","date":"2022-04-05","arxiv_id":"2204.02381","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-data-selection-via-discrete","title":"Unsupervised Data Selection via Discrete Speech Representation for ASR","date":"2022-04-05","arxiv_id":"2204.01981","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-gender-impact-in-self-supervised","title":"A Study of Gender Impact in Self-supervised Models for Speech-to-Text Systems","date":"2022-04-04","arxiv_id":"2204.01397","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-joint-speech-text-embeddings-for","title":"An Analysis of Semantically-Aligned Speech-Text Embeddings","date":"2022-04-04","arxiv_id":"2204.01235","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-self-supervised-speech","title":"Cross-lingual Self-Supervised Speech Representations for Improved Dysarthric Speech Recognition","date":"2022-04-04","arxiv_id":"2204.01670","repositories_listed":0,"syntology":null},{"url":null,"slug":"deliberation-model-for-on-device-spoken","title":"Deliberation Model for On-Device Spoken Language Understanding","date":"2022-04-04","arxiv_id":"2204.01893","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-model-for-named-entity-recognition","title":"End-to-end model for named entity recognition from speech without paired training data","date":"2022-04-02","arxiv_id":"2204.00803","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-real-time-personalized-speech","title":"Fast Real-time Personalized Speech Enhancement: End-to-End Enhancement Network (E3Net) and Knowledge Distillation","date":"2022-04-02","arxiv_id":"2204.00771","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-integration-of-speech-recognition","title":"End-to-End Integration of Speech Recognition, Speech Enhancement, and Self-Supervised Learning Representation","date":"2022-04-01","arxiv_id":"2204.00540","repositories_listed":0,"syntology":null}],"record_sha256":"a043ff8cefcc6f31852e67ecc13aaadc14e881670d8ebe7bf439342da8ecf068","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}