{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/14","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":31,"rows_per_page":100,"rows":[1301,1400],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/13","next":"/task/automatic-speech-recognition/papers/15","papers":[{"url":null,"slug":"speech-corpora-divergence-based-unsupervised","title":"Speech Corpora Divergence Based Unsupervised Data Selection for ASR","date":"2023-02-26","arxiv_id":"2302.13222","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-knowledge-distillation-of-self","title":"Ensemble knowledge distillation of self-supervised speech models","date":"2023-02-24","arxiv_id":"2302.12757","repositories_listed":0,"syntology":null},{"url":null,"slug":"factual-consistency-oriented-speech","title":"Factual Consistency Oriented Speech Recognition","date":"2023-02-24","arxiv_id":"2302.12369","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-automatic-speech-recognition-in-an","title":"Evaluating Automatic Speech Recognition in an Incremental Setting","date":"2023-02-23","arxiv_id":"2302.12049","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-contextual-spelling-correction-by","title":"Improving Contextual Spelling Correction by External Acoustics Attention and Semantic Aware Data Augmentation","date":"2023-02-22","arxiv_id":"2302.11192","repositories_listed":0,"syntology":null},{"url":null,"slug":"madi-inter-domain-matching-and-intra-domain","title":"MADI: Inter-domain Matching and Intra-domain Discrimination for Cross-domain Speech Recognition","date":"2023-02-22","arxiv_id":"2302.11224","repositories_listed":0,"syntology":null},{"url":null,"slug":"uml-a-universal-monolingual-output-layer-for","title":"UML: A Universal Monolingual Output Layer for Multilingual ASR","date":"2023-02-22","arxiv_id":"2302.11186","repositories_listed":0,"syntology":null},{"url":null,"slug":"connecting-humanities-and-social-sciences","title":"Connecting Humanities and Social Sciences: Applying Language and Speech Technology to Online Panel Surveys","date":"2023-02-21","arxiv_id":"2302.10593","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-asr-free-fluency-scoring-approach-with","title":"An ASR-free Fluency Scoring Approach with Self-Supervised Learning","date":"2023-02-20","arxiv_id":"2302.09928","repositories_listed":0,"syntology":null},{"url":null,"slug":"emphasizing-unseen-words-new-vocabulary","title":"Emphasizing Unseen Words: New Vocabulary Acquisition for End-to-End Speech Recognition","date":"2023-02-20","arxiv_id":"2302.09723","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-and-language-change-detection-using","title":"Speaker and Language Change Detection using Wav2vec2 and Whisper","date":"2023-02-18","arxiv_id":"2302.09381","repositories_listed":0,"syntology":null},{"url":null,"slug":"massively-multilingual-shallow-fusion-with","title":"Massively Multilingual Shallow Fusion with Large Language Models","date":"2023-02-17","arxiv_id":"2302.08917","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptable-end-to-end-asr-models-using","title":"Adaptable End-to-End ASR Models using Replaceable Internal LMs and Residual Softmax","date":"2023-02-16","arxiv_id":"2302.08579","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-axonal-delays-in-feedforward-spiking","slug":"adaptive-axonal-delays-in-feedforward-spiking","title":"Adaptive Axonal Delays in feedforward spiking neural networks for accurate spoken word recognition","date":"2023-02-16","arxiv_id":"2302.08607","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-change-detection-for-transformer","title":"Speaker Change Detection for Transformer Transducer ASR","date":"2023-02-16","arxiv_id":"2302.08549","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilising-and-accelerating-light-gated","title":"Stabilising and accelerating light gated recurrent units for automatic speech recognition","date":"2023-02-16","arxiv_id":"2302.10144","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-bundestag-a-large-scale-political-debate","title":"ASR Bundestag: A Large-Scale political debate dataset in German","date":"2023-02-12","arxiv_id":"2302.06008","repositories_listed":0,"syntology":null},{"url":null,"slug":"patcorrect-non-autoregressive-phoneme","title":"PATCorrect: Non-autoregressive Phoneme-augmented Transformer for ASR Error Correction","date":"2023-02-10","arxiv_id":"2302.05040","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-supplementary-text-data-to-kick","title":"Leveraging supplementary text data to kick-start automatic speech recognition system development with limited transcriptions","date":"2023-02-09","arxiv_id":"2302.04975","repositories_listed":0,"syntology":null},{"url":null,"slug":"pamp-a-unified-framework-boosting-low","title":"MAC: A unified framework boosting low resource automatic speech recognition","date":"2023-02-05","arxiv_id":"2302.03498","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rare-words-recognition-through","title":"Improving Rare Words Recognition through Homophone Extension and Unified Writing for Low-resource Cantonese Speech Recognition","date":"2023-02-02","arxiv_id":"2302.00836","repositories_listed":0,"syntology":null},{"url":null,"slug":"fillers-in-spoken-language-understanding","title":"Fillers in Spoken Language Understanding: Computational and Psycholinguistic Perspectives","date":"2023-01-25","arxiv_id":"2301.10761","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-purpose-audio-visual-corpus-for-multi","title":"A Multi-Purpose Audio-Visual Corpus for Multi-Modal Persian Speech Recognition: the Arman-AV Dataset","date":"2023-01-21","arxiv_id":"2301.10180","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-agnostic-data-driven-inverse-text","title":"Language Agnostic Data-Driven Inverse Text Normalization","date":"2023-01-20","arxiv_id":"2301.08506","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-english-to-more-languages-parameter","title":"From English to More Languages: Parameter-Efficient Model Reprogramming for Cross-Lingual Speech Recognition","date":"2023-01-19","arxiv_id":"2301.07851","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesspeech-a-bayesian-transformer-network","title":"BayesSpeech: A Bayesian Transformer Network for Automatic Speech Recognition","date":"2023-01-16","arxiv_id":"2301.11276","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-resolution-location-based-training-for","title":"Multi-resolution location-based training for multi-channel continuous speech separation","date":"2023-01-16","arxiv_id":"2301.06458","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-kaldi-for-automatic-speech-recognition","title":"Using Kaldi for Automatic Speech Recognition of Conversational Austrian German","date":"2023-01-16","arxiv_id":"2301.06475","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-punctuation-a-novel-punctuation","title":"Streaming Punctuation: A Novel Punctuation Technique Leveraging Bidirectional Context for Continuous Speech Recognition","date":"2023-01-10","arxiv_id":"2301.03819","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-lookup-dictionary-based","title":"Memory Augmented Lookup Dictionary based Language Modeling for Automatic Speech Recognition","date":"2022-12-30","arxiv_id":"2301.00066","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-be-so-sure-boosting-asr-decoding-via","title":"Don't Be So Sure! Boosting ASR Decoding via Confidence Relaxation","date":"2022-12-27","arxiv_id":"2212.13378","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-entropy-regularization","title":"Alignment Entropy Regularization","date":"2022-12-22","arxiv_id":"2212.12442","repositories_listed":0,"syntology":null},{"url":null,"slug":"4d-asr-joint-modeling-of-ctc-attention","title":"4D ASR: Joint modeling of CTC, Attention, Transducer, and Mask-Predict decoders","date":"2022-12-21","arxiv_id":"2212.10818","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-automatic-speech-recognition-model","title":"End-to-End Automatic Speech Recognition model for the Sudanese Dialect","date":"2022-12-21","arxiv_id":"2212.10826","repositories_listed":0,"syntology":null},{"url":null,"slug":"mu-2-slam-multitask-multilingual-speech-and","title":"Mu$^{2}$SLAM: Multitask, Multilingual Speech and Language Models","date":"2022-12-19","arxiv_id":"2212.09553","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-fine-tuning-of-self-supervised","title":"Context-aware Fine-tuning of Self-supervised Speech Models","date":"2022-12-16","arxiv_id":"2212.08542","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-entropy-based-methods-of-word-level","title":"Fast Entropy-Based Methods of Word-Level Confidence Estimation for End-To-End Automatic Speech Recognition","date":"2022-12-16","arxiv_id":"2212.08703","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-aware-dialog-system-technology","title":"Speech Aware Dialog System Technology Challenge (DSTC11)","date":"2022-12-16","arxiv_id":"2212.08704","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-fast-slow-encoder-based-transducer","title":"Improving Fast-slow Encoder based Transducer with Streaming Deliberation","date":"2022-12-15","arxiv_id":"2212.07650","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-prosody-representations-with","title":"Disentangling Prosody Representations with Unsupervised Speech Reconstruction","date":"2022-12-14","arxiv_id":"2212.06972","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-and-natural-language-processing","title":"Speech and Natural Language Processing Technologies for Pseudo-Pilot Simulator","date":"2022-12-14","arxiv_id":"2212.07164","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-translation-of-arabic-to","title":"End-to-End Speech Translation of Arabic to English Broadcast News","date":"2022-12-11","arxiv_id":"2212.05479","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-self-supervised-multilingual-speech","title":"Improved Self-Supervised Multilingual Speech Representation Learning Combined with Auxiliary Language Information","date":"2022-12-07","arxiv_id":"2212.03476","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-speech-pre-training-with-supervision","title":"Improved Speech Pre-Training with Supervision-Enhanced Acoustic Unit","date":"2022-12-07","arxiv_id":"2212.03482","repositories_listed":0,"syntology":null},{"url":null,"slug":"lattice-free-sequence-discriminative-training","title":"Lattice-Free Sequence Discriminative Training for Phoneme-Based Neural Transducers","date":"2022-12-07","arxiv_id":"2212.04325","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-multi-scale-self-supervised","title":"Progressive Multi-Scale Self-Supervised Learning for Speech Recognition","date":"2022-12-07","arxiv_id":"2212.03480","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-fine-tuning-data-selection-for","title":"Unsupervised Fine-Tuning Data Selection for ASR Using Self-Supervised Speech Models","date":"2022-12-03","arxiv_id":"2212.01661","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-for-on-device-speech","title":"Continual Learning for On-Device Speech Recognition using Disentangled Conformers","date":"2022-12-02","arxiv_id":"2212.01393","repositories_listed":0,"syntology":null},{"url":null,"slug":"preliminary-study-on-sscf-derived-polar","title":"Preliminary Study on SSCF-derived Polar Coordinate for ASR","date":"2022-11-30","arxiv_id":"2212.01245","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-transcription-of-uk-supreme-court","title":"Better Transcription of UK Supreme Court Hearings","date":"2022-11-29","arxiv_id":"2211.17094","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-reducing-the-distance-between","title":"Evaluating and reducing the distance between synthetic and real speech distributions","date":"2022-11-29","arxiv_id":"2211.16049","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-transducer-training-reduced-memory","title":"Neural Transducer Training: Reduced Memory Consumption with Sample-wise Computation","date":"2022-11-29","arxiv_id":"2211.16270","repositories_listed":0,"syntology":null},{"url":null,"slug":"inter-kd-intermediate-knowledge-distillation","title":"Inter-KD: Intermediate Knowledge Distillation for CTC-Based Automatic Speech Recognition","date":"2022-11-28","arxiv_id":"2211.15075","repositories_listed":0,"syntology":null},{"url":null,"slug":"bidirectional-representations-for-low","title":"Bidirectional Representations for Low Resource Spoken Language Understanding","date":"2022-11-24","arxiv_id":"2211.14320","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitask-learning-for-low-resource-spoken","title":"Multitask Learning for Low Resource Spoken Language Understanding","date":"2022-11-24","arxiv_id":"2211.13703","repositories_listed":0,"syntology":null},{"url":null,"slug":"device-directedness-with-contextual-cues-for","title":"Device Directedness with Contextual Cues for Spoken Dialog Systems","date":"2022-11-23","arxiv_id":"2211.13280","repositories_listed":0,"syntology":null},{"url":"/paper/benchmarking-evaluation-metrics-for-code","slug":"benchmarking-evaluation-metrics-for-code","title":"Benchmarking Evaluation Metrics for Code-Switching Automatic Speech Recognition","date":"2022-11-22","arxiv_id":"2211.16319","repositories_listed":0,"syntology":null},{"url":null,"slug":"complex-valued-time-frequency-self-attention","title":"Complex-Valued Time-Frequency Self-Attention for Speech Dereverberation","date":"2022-11-22","arxiv_id":"2211.12632","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequentially-sampled-chunk-conformer-or","title":"SSCFormer: Push the Limit of Chunk-wise Conformer for Streaming ASR Using Sequentially Sampled Chunks and Chunked Causal Convolution","date":"2022-11-21","arxiv_id":"2211.11419","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechnet-weakly-supervised-end-to-end-speech","title":"SpeechNet: Weakly Supervised, End-to-End Speech Recognition at Industrial Scale","date":"2022-11-21","arxiv_id":"2211.11740","repositories_listed":0,"syntology":null},{"url":null,"slug":"hey-asr-system-why-aren-t-you-more-inclusive","title":"Hey ASR System! Why Aren't You More Inclusive? Automatic Speech Recognition Systems' Bias and Proposed Bias Mitigation Techniques. A Literature Review","date":"2022-11-17","arxiv_id":"2211.09511","repositories_listed":0,"syntology":null},{"url":null,"slug":"longfnt-long-form-speech-recognition-with","title":"LongFNT: Long-form Speech Recognition with Factorized Neural Transducer","date":"2022-11-17","arxiv_id":"2211.09412","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-model-based-speaker-adaptation","title":"Unsupervised Model-based speaker adaptation of end-to-end lattice-free MMI model for speech recognition","date":"2022-11-17","arxiv_id":"2211.09313","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-with-unsupervised-speaking","title":"Improving Speech Emotion Recognition with Unsupervised Speaking Style Transfer","date":"2022-11-16","arxiv_id":"2211.08843","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-using-the-ua-speech-and-torgo-databases-to","title":"On using the UA-Speech and TORGO databases to validate automatic dysarthric speech classification approaches","date":"2022-11-16","arxiv_id":"2211.08833","repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-semantics-into-speech-encoders","title":"Introducing Semantics into Speech Encoders","date":"2022-11-15","arxiv_id":"2211.08402","repositories_listed":0,"syntology":null},{"url":null,"slug":"align-write-re-order-explainable-end-to-end","title":"Align, Write, Re-order: Explainable End-to-End Speech Translation via Operation Sequence Generation","date":"2022-11-11","arxiv_id":"2211.05967","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-trade-offs-in-speech-separation-with","title":"Handling Trade-Offs in Speech Separation with Sparsely-Gated Mixture of Experts","date":"2022-11-11","arxiv_id":"2211.06493","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-integration-of-pre-trained-ssl","title":"A Study on the Integration of Pre-trained SSL, ASR, LM and SLU Models for Spoken Language Understanding","date":"2022-11-10","arxiv_id":"2211.05869","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-multi-corpora-language-model","title":"Adaptive Multi-Corpora Language Model Training for Speech Recognition","date":"2022-11-09","arxiv_id":"2211.05121","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-noisy-student-training-on-non","title":"Improving Noisy Student Training on Non-target Domain Data for Automatic Speech Recognition","date":"2022-11-09","arxiv_id":"2211.04717","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-evaluation-of-a-spoken-dialogue","title":"End-to-End Evaluation of a Spoken Dialogue System for Learning Basic Mathematics","date":"2022-11-07","arxiv_id":"2211.03511","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-fast-and-accurate-on-device-inverse","title":"Streaming, fast and accurate on-device Inverse Text Normalization for Automatic Speech Recognition","date":"2022-11-07","arxiv_id":"2211.03721","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-speech-and-textual-pre-trained","title":"Bridging Speech and Textual Pre-trained Models with Unsupervised ASR","date":"2022-11-06","arxiv_id":"2211.03025","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-automated-speech-recognition","title":"Evaluation of Automated Speech Recognition Systems for Conversational Speech: A Linguistic Perspective","date":"2022-11-05","arxiv_id":"2211.02812","repositories_listed":0,"syntology":null},{"url":null,"slug":"lamassu-streaming-language-agnostic","title":"LAMASSU: Streaming Language-Agnostic Multilingual Speech Recognition and Translation Using Neural Transducers","date":"2022-11-05","arxiv_id":"2211.02809","repositories_listed":0,"syntology":null},{"url":null,"slug":"biased-self-supervised-learning-for-asr","title":"Biased Self-supervised learning for ASR","date":"2022-11-04","arxiv_id":"2211.02536","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-transfer-learning-from","title":"Resource-Efficient Transfer Learning From Speech Foundation Model Using Hierarchical Feature Fusion","date":"2022-11-04","arxiv_id":"2211.02712","repositories_listed":0,"syntology":null},{"url":null,"slug":"stutter-tts-controlled-synthesis-and-improved","title":"Stutter-TTS: Controlled Synthesis and Improved Recognition of Stuttered Speech","date":"2022-11-04","arxiv_id":"2211.09731","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-sd-text-h-text-sd-a-new-hybrid","title":"H_eval: A new hybrid evaluation metric for automatic speech recognition tasks","date":"2022-11-03","arxiv_id":"2211.01722","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-domain-features-for-detecting","title":"Leveraging Domain Features for Detecting Adversarial Attacks Against Deep Speech Recognition in Noise","date":"2022-11-03","arxiv_id":"2211.01621","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-assisted-multi-target-units-modeling","title":"Phonetic-assisted Multi-Target Units Modeling for Improving Conformer-Transducer ASR system","date":"2022-11-03","arxiv_id":"2211.01571","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-statistical-representations-for-end","title":"Probing Statistical Representations For End-To-End ASR","date":"2022-11-03","arxiv_id":"2211.01993","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-audio-visual-speech-recognition","title":"Streaming Audio-Visual Speech Recognition with Alignment Regularization","date":"2022-11-03","arxiv_id":"2211.02133","repositories_listed":0,"syntology":null},{"url":null,"slug":"bectra-transducer-based-end-to-end-asr-with","title":"BECTRA: Transducer-based End-to-End ASR with BERT-Enhanced Encoder","date":"2022-11-02","arxiv_id":"2211.00792","repositories_listed":0,"syntology":null},{"url":null,"slug":"monolingual-recognizers-fusion-for-code","title":"Monolingual Recognizers Fusion for Code-switching Speech Recognition","date":"2022-11-02","arxiv_id":"2211.01046","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-speaking-or-more-speakers","title":"More Speaking or More Speakers?","date":"2022-11-02","arxiv_id":"2211.00854","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-code-switched-speech","title":"Towards Zero-Shot Code-Switched Speech Recognition","date":"2022-11-02","arxiv_id":"2211.01458","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-multichannel-speaker","title":"A Comparative Study on Multichannel Speaker-Attributed Automatic Speech Recognition in Multi-party Meetings","date":"2022-11-01","arxiv_id":"2211.00511","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-study-on-automated-speaking","title":"A Preliminary Study on Automated Speaking Assessment of English as a Second Language (ESL) Students","date":"2022-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-self-supervised-models-to-multi","title":"Adapting self-supervised models to multi-talker speech recognition using speaker embeddings","date":"2022-11-01","arxiv_id":"2211.00482","repositories_listed":0,"syntology":null},{"url":null,"slug":"mandarin-english-code-switching-speech-1","title":"Mandarin-English Code-Switching Speech Recognition System for Specific Domain","date":"2022-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-end-to-end-speech-recognition-and","title":"Unified End-to-End Speech Recognition and Endpointing for Fast and Efficient Speech Systems","date":"2022-11-01","arxiv_id":"2211.00786","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-degenerating-speech-due-to","title":"An analysis of degenerating speech due to progressive dysarthria on ASR performance","date":"2022-10-31","arxiv_id":"2211.00089","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-and","title":"Audio-Visual Speech Enhancement and Separation by Utilizing Multi-Modal Self-Supervised Embeddings","date":"2022-10-31","arxiv_id":"2210.17456","repositories_listed":0,"syntology":null},{"url":null,"slug":"diacorrect-end-to-end-error-correction-for","title":"DiaCorrect: End-to-end error correction for speaker diarization","date":"2022-10-31","arxiv_id":"2210.17189","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusionformer-fusing-operations-in-transformer","title":"FusionFormer: Fusing Operations in Transformer for Efficient Streaming Speech Recognition","date":"2022-10-31","arxiv_id":"2210.17079","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-state-space-decoder-for-speech","title":"Structured State Space Decoder for Speech Recognition and Synthesis","date":"2022-10-31","arxiv_id":"2210.17098","repositories_listed":0,"syntology":null},{"url":null,"slug":"dude-dual-decoder-multilingual-asr-for-indian","title":"DuDe: Dual-Decoder Multilingual ASR for Indian Languages using Common Label Set","date":"2022-10-30","arxiv_id":"2210.16739","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonemic-representation-and-transcription-for","title":"Phonemic Representation and Transcription for Speech to Text Applications for Under-resourced Indigenous African Languages: The Case of Kiswahili","date":"2022-10-29","arxiv_id":"2210.16537","repositories_listed":0,"syntology":null}],"record_sha256":"e2adcc55d1652029e6a58949684c8105c30c5e69d8a32453ed3cf258e49fc4b2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}