{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/16","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":16,"pages_in_order":32,"rows_per_page":100,"rows":[1501,1600],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/15","next":"/task/automatic-speech-recognition-2/papers/17","papers":[{"url":null,"slug":"end-to-end-spoken-language-understanding-4","title":"End-to-end spoken language understanding using joint CTC loss and self-supervised, pretrained acoustic encoders","date":"2023-05-04","arxiv_id":"2305.02937","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-transducer-and-attention-based-encoder","title":"Hybrid Transducer and Attention based Encoder-Decoder Modeling for Speech-to-Text Tasks","date":"2023-05-04","arxiv_id":"2305.03101","repositories_listed":0,"syntology":null},{"url":null,"slug":"considerations-for-ethical-speech-recognition","title":"Considerations for Ethical Speech Recognition Datasets","date":"2023-05-03","arxiv_id":"2305.02081","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-integration-of-pipeline-and","title":"A Study on the Integration of Pipeline and E2E SLU systems for Spoken Semantic Parsing toward STOP Quality Challenge","date":"2023-05-02","arxiv_id":"2305.01620","repositories_listed":0,"syntology":null},{"url":null,"slug":"lessons-learned-in-atco2-5000-hours-of-air","title":"Lessons Learned in ATCO2: 5000 hours of Air Traffic Control Communications for Robust Automatic Speech Recognition and Understanding","date":"2023-05-02","arxiv_id":"2305.01155","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-learning-techniques-for-3","title":"A Review of Deep Learning Techniques for Speech Processing","date":"2023-04-30","arxiv_id":"2305.00359","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-non-native-speech-corpus-featuring","title":"Building a Non-native Speech Corpus Featuring Chinese-English Bilingual Children: Compilation and Rationale","date":"2023-04-30","arxiv_id":"2305.00446","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-transfer-learning-for-automatic-speech","title":"Deep Transfer Learning for Automatic Speech Recognition: Towards Better Generalization","date":"2023-04-27","arxiv_id":"2304.14535","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-shared-speech-text","title":"Understanding Shared Speech-Text Representations","date":"2023-04-27","arxiv_id":"2304.14514","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-regularised-minimum-latency-training-for","title":"Self-regularised Minimum Latency Training for Streaming Transformer-based Speech Recognition","date":"2023-04-24","arxiv_id":"2304.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-end-to-end-approaches-for","title":"Non-autoregressive End-to-end Approaches for Joint Automatic Speech Recognition and Spoken Language Understanding","date":"2023-04-21","arxiv_id":"2304.10869","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-universal-defense-for-query-based","title":"Towards the Universal Defense for Query-Based Audio Adversarial Attacks","date":"2023-04-20","arxiv_id":"2304.10088","repositories_listed":0,"syntology":null},{"url":null,"slug":"security-and-privacy-problems-in-voice","title":"Security and Privacy Problems in Voice Assistant Applications: A Survey","date":"2023-04-19","arxiv_id":"2304.09486","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-short-video-rumor-detection-system","title":"Multimodal Short Video Rumor Detection System Based on Contrastive Learning","date":"2023-04-17","arxiv_id":"2304.08401","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-virtual-simulation-pilot-agent-for-training","title":"A Virtual Simulation-Pilot Agent for Training of Air Traffic Controllers","date":"2023-04-16","arxiv_id":"2304.07842","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-ctc-alignment-based-non-autoregressive","title":"A CTC Alignment-based Non-autoregressive Transformer for End-to-end Automatic Speech Recognition","date":"2023-04-15","arxiv_id":"2304.07611","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-speaker-anonymization-on","title":"Evaluation of Speaker Anonymization on Emotional Speech","date":"2023-04-15","arxiv_id":"2305.01759","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-document-grounded-dialog","title":"Task-oriented Document-Grounded Dialog Systems by HLTPR@RWTH for DSTC9 and DSTC10","date":"2023-04-14","arxiv_id":"2304.07101","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-contrastive-predictive-coding","title":"Regularizing Contrastive Predictive Coding for Speech Applications","date":"2023-04-12","arxiv_id":"2304.05974","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-reconstruction-from-silent-tongue-and","title":"Speech Reconstruction from Silent Tongue and Lip Articulation By Pseudo Target Generation and Domain Adversarial Training","date":"2023-04-12","arxiv_id":"2304.05574","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2code-restore-clean-speech-representations","title":"Wav2code: Restore Clean Speech Representations via Codebook Lookup for Noise-Robust ASR","date":"2023-04-11","arxiv_id":"2304.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-and-accurate-self-supervised","title":"Scalable and Accurate Self-supervised Multimodal Representation Learning without Aligned Video and Text Data","date":"2023-04-04","arxiv_id":"2304.02080","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-based-source","title":"Self-Supervised Learning-Based Source Separation for Meeting Data","date":"2023-04-03","arxiv_id":"2304.00871","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-word-error-rate-estimation-e","title":"Multilingual Word Error Rate Estimation: e-WER3","date":"2023-04-02","arxiv_id":"2304.00649","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialog-act-guided-contextual-adapter-for","title":"Dialog act guided contextual adapter for personalized speech recognition","date":"2023-03-31","arxiv_id":"2303.17799","repositories_listed":0,"syntology":null},{"url":"/paper/improving-the-previous-state-of-the-art","slug":"improving-the-previous-state-of-the-art","title":"Improving the previous state-of-the-art Frisian ASR by fine-tuning XLS-R","date":"2023-03-31","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/the-edinburgh-international-accents-of","slug":"the-edinburgh-international-accents-of","title":"The Edinburgh International Accents of English Corpus: Towards the Democratization of English ASR","date":"2023-03-31","arxiv_id":"2303.18110","repositories_listed":0,"syntology":null},{"url":null,"slug":"procter-pronunciation-aware-contextual","title":"PROCTER: PROnunciation-aware ConTextual adaptER for personalized speech recognition in neural transducers","date":"2023-03-30","arxiv_id":"2303.17131","repositories_listed":0,"syntology":null},{"url":null,"slug":"avformer-injecting-vision-into-frozen-speech","title":"AVFormer: Injecting Vision into Frozen Speech Models for Zero-Shot AV-ASR","date":"2023-03-29","arxiv_id":"2303.16501","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-unsupervised-and-supervised-learning","title":"Joint unsupervised and supervised learning for context-aware language identification","date":"2023-03-29","arxiv_id":"2303.16511","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-is-all-you-need-personalizing-asr-models","title":"Text is All You Need: Personalizing ASR Models using Controllable Speech Synthesis","date":"2023-03-27","arxiv_id":"2303.14885","repositories_listed":0,"syntology":null},{"url":"/paper/beyond-universal-transformer-block-reusing","slug":"beyond-universal-transformer-block-reusing","title":"Beyond Universal Transformer: block reusing with adaptor in Transformer for automatic speech recognition","date":"2023-03-23","arxiv_id":"2303.13072","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-unsupervised-speech-recognition","title":"Enhancing Unsupervised Speech Recognition with Diffusion GANs","date":"2023-03-23","arxiv_id":"2303.13559","repositories_listed":0,"syntology":null},{"url":null,"slug":"pyramid-multi-branch-fusion-dcnn-with-multi","title":"Pyramid Multi-branch Fusion DCNN with Multi-Head Self-Attention for Mandarin Speech Recognition","date":"2023-03-23","arxiv_id":"2303.13243","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-with-speech","title":"Self-supervised Learning with Speech Modulation Dropout","date":"2023-03-22","arxiv_id":"2303.12908","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-integration-of-speech-separation","title":"End-to-End Integration of Speech Separation and Voice Activity Detection for Low-Latency Diarization of Telephone Conversations","date":"2023-03-21","arxiv_id":"2303.12002","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-in-speech-processing-a-survey","title":"Transformers in Speech Processing: A Survey","date":"2023-03-21","arxiv_id":"2303.11607","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switching-text-generation-and-injection","title":"Code-Switching Text Generation and Injection in Mandarin-English ASR","date":"2023-03-20","arxiv_id":"2303.10949","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-from-multiple","title":"Knowledge Distillation from Multiple Foundation Models for End-to-End Speech Recognition","date":"2023-03-20","arxiv_id":"2303.10917","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-learning-system-for-domain-specific","title":"A Deep Learning System for Domain-specific Speech Recognition","date":"2023-03-18","arxiv_id":"2303.10510","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillw2v2-a-small-and-streaming-wav2vec-2-0","title":"DistillW2V2: A Small and Streaming Wav2vec 2.0 Based ASR Model","date":"2023-03-16","arxiv_id":"2303.09278","repositories_listed":0,"syntology":null},{"url":null,"slug":"trustera-a-live-conversation-redaction-system","title":"Trustera: A Live Conversation Redaction System","date":"2023-03-16","arxiv_id":"2303.09438","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-information-matters-for-asr-error","title":"Visual Information Matters for ASR Error Correction","date":"2023-03-16","arxiv_id":"2303.10160","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accented-speech-recognition-with","title":"Improving Accented Speech Recognition with Multi-Domain Training","date":"2023-03-14","arxiv_id":"2303.07924","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-intent-classification-accuracy","title":"Improving the Intent Classification accuracy in Noisy Environment","date":"2023-03-12","arxiv_id":"2303.06585","repositories_listed":0,"syntology":null},{"url":null,"slug":"clinical-bertscore-an-improved-measure-of","title":"Clinical BERTScore: An Improved Measure of Automatic Speech Recognition Performance in Clinical Settings","date":"2023-03-10","arxiv_id":"2303.05737","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixpgd-hybrid-adversarial-training-for-speech","title":"MIXPGD: Hybrid Adversarial Training for Speech Recognition Systems","date":"2023-03-10","arxiv_id":"2303.05758","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2vec-and-its-current-potential-to","title":"wav2vec and its current potential to Automatic Speech Recognition in German for the usage in Digital History: A comparative assessment of available ASR-technologies for the use in cultural heritage contexts","date":"2023-03-06","arxiv_id":"2303.06026","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-recognition-a-survey","title":"End-to-End Speech Recognition: A Survey","date":"2023-03-03","arxiv_id":"2303.03329","repositories_listed":0,"syntology":null},{"url":null,"slug":"google-usm-scaling-automatic-speech","title":"Google USM: Scaling Automatic Speech Recognition Beyond 100 Languages","date":"2023-03-02","arxiv_id":"2303.01037","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-text-corpora-for-end-to-end","title":"Leveraging Large Text Corpora for End-to-End Speech Summarization","date":"2023-03-02","arxiv_id":"2303.00978","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-redundancy-in-multiple-audio","title":"Leveraging Redundancy in Multiple Audio Signals for Far-Field Speech Recognition","date":"2023-03-01","arxiv_id":"2303.00692","repositories_listed":0,"syntology":null},{"url":null,"slug":"n-best-t5-robust-asr-error-correction-using","title":"N-best T5: Robust ASR Error Correction using Multiple Input Hypotheses and Constrained Decoding Space","date":"2023-03-01","arxiv_id":"2303.00456","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-cross-accent-data-augmentation-for","title":"Synthetic Cross-accent Data Augmentation for Automatic Speech Recognition","date":"2023-03-01","arxiv_id":"2303.00802","repositories_listed":0,"syntology":null},{"url":null,"slug":"practice-of-the-conformer-enhanced-audio","title":"Practice of the conformer enhanced AUDIO-VISUAL HUBERT on Mandarin and English","date":"2023-02-28","arxiv_id":"2303.12187","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-speech-data-augmentation","title":"A Comparison of Speech Data Augmentation Methods Using S3PRL Toolkit","date":"2023-02-27","arxiv_id":"2303.00510","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-visual-forced-alignment-learning-to","title":"Deep Visual Forced Alignment: Learning to Align Transcription with Talking Face Video","date":"2023-02-27","arxiv_id":"2303.08670","repositories_listed":0,"syntology":null},{"url":null,"slug":"diacritic-recognition-performance-in-arabic","title":"Diacritic Recognition Performance in Arabic ASR","date":"2023-02-27","arxiv_id":"2302.14022","repositories_listed":0,"syntology":null},{"url":null,"slug":"explanations-for-automatic-speech-recognition","title":"Explanations for Automatic Speech Recognition","date":"2023-02-27","arxiv_id":"2302.14062","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-medical-speech-to-text-accuracy","title":"Improving Medical Speech-to-Text Accuracy with Vision-Language Pre-training Model","date":"2023-02-27","arxiv_id":"2303.00091","repositories_listed":0,"syntology":null},{"url":null,"slug":"mole-mixture-of-language-experts-for-multi","title":"MoLE : Mixture of Language Experts for Multi-Lingual Automatic Speech Recognition","date":"2023-02-27","arxiv_id":"2302.13750","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-corpora-divergence-based-unsupervised","title":"Speech Corpora Divergence Based Unsupervised Data Selection for ASR","date":"2023-02-26","arxiv_id":"2302.13222","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-knowledge-distillation-of-self","title":"Ensemble knowledge distillation of self-supervised speech models","date":"2023-02-24","arxiv_id":"2302.12757","repositories_listed":0,"syntology":null},{"url":null,"slug":"factual-consistency-oriented-speech","title":"Factual Consistency Oriented Speech Recognition","date":"2023-02-24","arxiv_id":"2302.12369","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-automatic-speech-recognition-in-an","title":"Evaluating Automatic Speech Recognition in an Incremental Setting","date":"2023-02-23","arxiv_id":"2302.12049","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-contextual-spelling-correction-by","title":"Improving Contextual Spelling Correction by External Acoustics Attention and Semantic Aware Data Augmentation","date":"2023-02-22","arxiv_id":"2302.11192","repositories_listed":0,"syntology":null},{"url":null,"slug":"madi-inter-domain-matching-and-intra-domain","title":"MADI: Inter-domain Matching and Intra-domain Discrimination for Cross-domain Speech Recognition","date":"2023-02-22","arxiv_id":"2302.11224","repositories_listed":0,"syntology":null},{"url":null,"slug":"uml-a-universal-monolingual-output-layer-for","title":"UML: A Universal Monolingual Output Layer for Multilingual ASR","date":"2023-02-22","arxiv_id":"2302.11186","repositories_listed":0,"syntology":null},{"url":null,"slug":"connecting-humanities-and-social-sciences","title":"Connecting Humanities and Social Sciences: Applying Language and Speech Technology to Online Panel Surveys","date":"2023-02-21","arxiv_id":"2302.10593","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-asr-free-fluency-scoring-approach-with","title":"An ASR-free Fluency Scoring Approach with Self-Supervised Learning","date":"2023-02-20","arxiv_id":"2302.09928","repositories_listed":0,"syntology":null},{"url":null,"slug":"emphasizing-unseen-words-new-vocabulary","title":"Emphasizing Unseen Words: New Vocabulary Acquisition for End-to-End Speech Recognition","date":"2023-02-20","arxiv_id":"2302.09723","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-and-language-change-detection-using","title":"Speaker and Language Change Detection using Wav2vec2 and Whisper","date":"2023-02-18","arxiv_id":"2302.09381","repositories_listed":0,"syntology":null},{"url":null,"slug":"massively-multilingual-shallow-fusion-with","title":"Massively Multilingual Shallow Fusion with Large Language Models","date":"2023-02-17","arxiv_id":"2302.08917","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-end-to-end-asr-models-using","title":"Adaptable End-to-End ASR Models using Replaceable Internal LMs and Residual Softmax","date":"2023-02-16","arxiv_id":"2302.08579","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-axonal-delays-in-feedforward-spiking","slug":"adaptive-axonal-delays-in-feedforward-spiking","title":"Adaptive Axonal Delays in feedforward spiking neural networks for accurate spoken word recognition","date":"2023-02-16","arxiv_id":"2302.08607","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-change-detection-for-transformer","title":"Speaker Change Detection for Transformer Transducer ASR","date":"2023-02-16","arxiv_id":"2302.08549","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilising-and-accelerating-light-gated","title":"Stabilising and accelerating light gated recurrent units for automatic speech recognition","date":"2023-02-16","arxiv_id":"2302.10144","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-bundestag-a-large-scale-political-debate","title":"ASR Bundestag: A Large-Scale political debate dataset in German","date":"2023-02-12","arxiv_id":"2302.06008","repositories_listed":0,"syntology":null},{"url":null,"slug":"patcorrect-non-autoregressive-phoneme","title":"PATCorrect: Non-autoregressive Phoneme-augmented Transformer for ASR Error Correction","date":"2023-02-10","arxiv_id":"2302.05040","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-supplementary-text-data-to-kick","title":"Leveraging supplementary text data to kick-start automatic speech recognition system development with limited transcriptions","date":"2023-02-09","arxiv_id":"2302.04975","repositories_listed":0,"syntology":null},{"url":null,"slug":"pamp-a-unified-framework-boosting-low","title":"MAC: A unified framework boosting low resource automatic speech recognition","date":"2023-02-05","arxiv_id":"2302.03498","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rare-words-recognition-through","title":"Improving Rare Words Recognition through Homophone Extension and Unified Writing for Low-resource Cantonese Speech Recognition","date":"2023-02-02","arxiv_id":"2302.00836","repositories_listed":0,"syntology":null},{"url":null,"slug":"fillers-in-spoken-language-understanding","title":"Fillers in Spoken Language Understanding: Computational and Psycholinguistic Perspectives","date":"2023-01-25","arxiv_id":"2301.10761","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-purpose-audio-visual-corpus-for-multi","title":"A Multi-Purpose Audio-Visual Corpus for Multi-Modal Persian Speech Recognition: the Arman-AV Dataset","date":"2023-01-21","arxiv_id":"2301.10180","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-agnostic-data-driven-inverse-text","title":"Language Agnostic Data-Driven Inverse Text Normalization","date":"2023-01-20","arxiv_id":"2301.08506","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-english-to-more-languages-parameter","title":"From English to More Languages: Parameter-Efficient Model Reprogramming for Cross-Lingual Speech Recognition","date":"2023-01-19","arxiv_id":"2301.07851","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesspeech-a-bayesian-transformer-network","title":"BayesSpeech: A Bayesian Transformer Network for Automatic Speech Recognition","date":"2023-01-16","arxiv_id":"2301.11276","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-resolution-location-based-training-for","title":"Multi-resolution location-based training for multi-channel continuous speech separation","date":"2023-01-16","arxiv_id":"2301.06458","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-kaldi-for-automatic-speech-recognition","title":"Using Kaldi for Automatic Speech Recognition of Conversational Austrian German","date":"2023-01-16","arxiv_id":"2301.06475","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-punctuation-a-novel-punctuation","title":"Streaming Punctuation: A Novel Punctuation Technique Leveraging Bidirectional Context for Continuous Speech Recognition","date":"2023-01-10","arxiv_id":"2301.03819","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-pre-training-for-vietnamese","title":"Unsupervised Pre-Training for Vietnamese Automatic Speech Recognition in the HYKIST Project","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-lookup-dictionary-based","title":"Memory Augmented Lookup Dictionary based Language Modeling for Automatic Speech Recognition","date":"2022-12-30","arxiv_id":"2301.00066","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-be-so-sure-boosting-asr-decoding-via","title":"Don't Be So Sure! Boosting ASR Decoding via Confidence Relaxation","date":"2022-12-27","arxiv_id":"2212.13378","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-entropy-regularization","title":"Alignment Entropy Regularization","date":"2022-12-22","arxiv_id":"2212.12442","repositories_listed":0,"syntology":null},{"url":null,"slug":"4d-asr-joint-modeling-of-ctc-attention","title":"4D ASR: Joint modeling of CTC, Attention, Transducer, and Mask-Predict decoders","date":"2022-12-21","arxiv_id":"2212.10818","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-automatic-speech-recognition-model","title":"End-to-End Automatic Speech Recognition model for the Sudanese Dialect","date":"2022-12-21","arxiv_id":"2212.10826","repositories_listed":0,"syntology":null},{"url":null,"slug":"mu-2-slam-multitask-multilingual-speech-and","title":"Mu$^{2}$SLAM: Multitask, Multilingual Speech and Language Models","date":"2022-12-19","arxiv_id":"2212.09553","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-fine-tuning-of-self-supervised","title":"Context-aware Fine-tuning of Self-supervised Speech Models","date":"2022-12-16","arxiv_id":"2212.08542","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-entropy-based-methods-of-word-level","title":"Fast Entropy-Based Methods of Word-Level Confidence Estimation for End-To-End Automatic Speech Recognition","date":"2022-12-16","arxiv_id":"2212.08703","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-aware-dialog-system-technology","title":"Speech Aware Dialog System Technology Challenge (DSTC11)","date":"2022-12-16","arxiv_id":"2212.08704","repositories_listed":0,"syntology":null}],"record_sha256":"dafedc857ea1cc4fb5cb1422ca8699634505455a0d3e854055a5024773f94ce2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}