{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/25","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":65,"rows_per_page":100,"rows":[2401,2500],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/24","next":"/task/speech-recognition/papers/26","papers":[{"url":null,"slug":"adapting-text-based-dialogue-state-tracker","title":"Adapting Text-based Dialogue State Tracker for Spoken Dialogues","date":"2023-08-29","arxiv_id":"2308.15053","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-approaches-to-spoken-content-embedding","title":"Neural approaches to spoken content embedding","date":"2023-08-28","arxiv_id":"2308.14905","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ustc-nercslip-systems-for-the-chime-7","title":"The USTC-NERCSLIP Systems for the CHiME-7 DASR Challenge","date":"2023-08-28","arxiv_id":"2308.14638","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-active-learning-optimizing","title":"Unsupervised Active Learning: Optimizing Labeling Cost-Effectiveness for Automatic Speech Recognition","date":"2023-08-28","arxiv_id":"2308.14814","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupled-structure-for-improved-adaptability","title":"Decoupled Structure for Improved Adaptability of End-to-End Models","date":"2023-08-25","arxiv_id":"2308.13345","repositories_listed":0,"syntology":null},{"url":"/paper/real-time-detection-of-ai-generated-speech","slug":"real-time-detection-of-ai-generated-speech","title":"Real-time Detection of AI-Generated Speech for DeepFake Voice Conversion","date":"2023-08-24","arxiv_id":"2308.12734","repositories_listed":0,"syntology":null},{"url":null,"slug":"adverb-visually-guided-audio-dereverberation","title":"AdVerb: Visually Guided Audio Dereverberation","date":"2023-08-23","arxiv_id":"2308.12370","repositories_listed":0,"syntology":null},{"url":null,"slug":"kinspeak-improving-speech-recognition-for","title":"KinSPEAK: Improving speech recognition for Kinyarwanda via semi-supervised learning methods","date":"2023-08-23","arxiv_id":"2308.11863","repositories_listed":0,"syntology":null},{"url":null,"slug":"convoifilter-a-case-study-of-doing-cocktail","title":"Convoifilter: A case study of doing cocktail party speech recognition","date":"2023-08-22","arxiv_id":"2308.11380","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-depression-related-topics-in","title":"Identifying depression-related topics in smartphone-collected free-response speech recordings using an automatic speech recognition system and a deep learning topic model","date":"2023-08-22","arxiv_id":"2308.11773","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-continuous-sign-language-2","title":"Improving Continuous Sign Language Recognition with Cross-Lingual Signs","date":"2023-08-21","arxiv_id":"2308.10809","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokensplit-using-discrete-speech","title":"TokenSplit: Using Discrete Speech Representations for Direct, Refined, and Transcript-Conditioned Speech Separation and Recognition","date":"2023-08-21","arxiv_id":"2308.10415","repositories_listed":0,"syntology":null},{"url":"/paper/another-point-of-view-on-visual-speech","slug":"another-point-of-view-on-visual-speech","title":"Another Point of View on Visual Speech Recognition","date":"2023-08-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"indonesian-automatic-speech-recognition-with","title":"Indonesian Automatic Speech Recognition with XLSR-53","date":"2023-08-20","arxiv_id":"2308.11589","repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-synthesis-of-dysarthric-speech-for","title":"Accurate synthesis of Dysarthric Speech for ASR data augmentation","date":"2023-08-16","arxiv_id":"2308.08438","repositories_listed":0,"syntology":null},{"url":null,"slug":"radio2text-streaming-speech-recognition-using","title":"Radio2Text: Streaming Speech Recognition Using mmWave Radio Signals","date":"2023-08-16","arxiv_id":"2308.08125","repositories_listed":0,"syntology":null},{"url":null,"slug":"akvsr-audio-knowledge-empowered-visual-speech","title":"AKVSR: Audio Knowledge Empowered Visual Speech Recognition by Compressing Audio Knowledge of a Pretrained Model","date":"2023-08-15","arxiv_id":"2308.07593","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-ctc-aed-model-with-integrated-ctc","title":"Improving CTC-AED model with integrated-CTC and auxiliary loss regularization","date":"2023-08-15","arxiv_id":"2308.08449","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-attribute-matrix-factorization-model","title":"Cross-Attribute Matrix Factorization Model with Shared User Embedding","date":"2023-08-14","arxiv_id":"2308.07284","repositories_listed":0,"syntology":null},{"url":null,"slug":"o-1-self-training-with-oracle-and-1-best","title":"O-1: Self-training with Oracle and 1-best Hypothesis","date":"2023-08-14","arxiv_id":"2308.07486","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-injection-for-capitalization-and-turn","title":"Text Injection for Capitalization and Turn-Taking Prediction in Speech Models","date":"2023-08-14","arxiv_id":"2308.07395","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-text-injection-to-improve-recognition","title":"Using Text Injection to Improve Recognition of Personal Identifiers in Speech","date":"2023-08-14","arxiv_id":"2308.07393","repositories_listed":0,"syntology":null},{"url":null,"slug":"alternative-pseudo-labeling-for-semi","title":"Alternative Pseudo-Labeling for Semi-Supervised Automatic Speech Recognition","date":"2023-08-12","arxiv_id":"2308.06547","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilingual-streaming-asr-with-grapheme-units","title":"Bilingual Streaming ASR with Grapheme units and Auxiliary Monolingual Loss","date":"2023-08-11","arxiv_id":"2308.06327","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-joint-speech-text-representations","title":"Improving Joint Speech-Text Representations Without Alignment","date":"2023-08-11","arxiv_id":"2308.06125","repositories_listed":0,"syntology":null},{"url":"/paper/lip2vec-efficient-and-robust-visual-speech","slug":"lip2vec-efficient-and-robust-visual-speech","title":"Lip2Vec: Efficient and Robust Visual Speech Recognition via Latent-to-Latent Visual to Audio Representation Mapping","date":"2023-08-11","arxiv_id":"2308.06112","repositories_listed":0,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":16,"n_pointer_only":19,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 1 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lip2vec-efficient-and-robust-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2308.06112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06112"}},"official":null}},{"url":null,"slug":"a-novel-self-training-approach-for-low","title":"A Novel Self-training Approach for Low-resource Speech Recognition","date":"2023-08-10","arxiv_id":"2308.05269","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-method-for-improving-accuracy-in","title":"A Novel Method for improving accuracy in neural network by reinstating traditional back propagation technique","date":"2023-08-09","arxiv_id":"2308.05059","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-resource-aware-structured-pruning-for","title":"FPGA Resource-aware Structured Pruning for Real-Time Neural Networks","date":"2023-08-09","arxiv_id":"2308.05170","repositories_listed":0,"syntology":null},{"url":null,"slug":"tssr-a-truncated-and-signed-square-root","title":"TSSR: A Truncated and Signed Square Root Activation Function for Neural Networks","date":"2023-08-09","arxiv_id":"2308.04832","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-out-of-distribution-dialect","title":"Unsupervised Out-of-Distribution Dialect Detection with Mahalanobis Distance","date":"2023-08-09","arxiv_id":"2308.04886","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-the-wav2vec-2-0","title":"Comparative Analysis of the wav2vec 2.0 Feature Extractor","date":"2023-08-08","arxiv_id":"2308.04286","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-chinese-asr-error-correction-with","title":"Boosting Chinese ASR Error Correction with Dynamic Error Scaling Mechanism","date":"2023-08-07","arxiv_id":"2308.03423","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogue-systems-can-generate-appropriate","title":"Dialogue Systems Can Generate Appropriate Responses without the Use of Question Marks? -- Investigation of the Effects of Question Marks on Dialogue Systems","date":"2023-08-07","arxiv_id":"2308.03293","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-critical-review-of-physics-informed-machine","title":"A Critical Review of Physics-Informed Machine Learning Applications in Subsurface Energy Systems","date":"2023-08-06","arxiv_id":"2308.04457","repositories_listed":0,"syntology":null},{"url":null,"slug":"approbivt-lead-asr-models-to-generalize","title":"ApproBiVT: Lead ASR Models to Generalize Better Using Approximated Bias-Variance Tradeoff Guided Early Stopping and Checkpoint Averaging","date":"2023-08-05","arxiv_id":"2308.02870","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-diarization-of-scripted-audiovisual","title":"Speaker Diarization of Scripted Audiovisual Content","date":"2023-08-04","arxiv_id":"2308.02160","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-representation-learning-for","title":"Federated Representation Learning for Automatic Speech Recognition","date":"2023-08-03","arxiv_id":"2308.02013","repositories_listed":0,"syntology":null},{"url":null,"slug":"careful-whisper-leveraging-advances-in","title":"Careful Whisper -- leveraging advances in automatic speech recognition for robust and interpretable aphasia subtype classification","date":"2023-08-02","arxiv_id":"2308.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-grapheme-to-phoneme-conversion-by","title":"Improving grapheme-to-phoneme conversion by learning pronunciations from speech recordings","date":"2023-07-31","arxiv_id":"2307.16643","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-end-to-end-asr-models-with","title":"Pre-training End-to-end ASR Models with Augmented Speech Samples Queried by Text","date":"2023-07-30","arxiv_id":"2307.16332","repositories_listed":0,"syntology":null},{"url":null,"slug":"unibrivl-robust-universal-representation-and","title":"UniBriVL: Robust Universal Representation and Generation of Audio Driven Diffusion Models","date":"2023-07-29","arxiv_id":"2307.15898","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-timing-bottleneck-why-timing-and-overlap","title":"The timing bottleneck: Why timing and overlap are mission-critical for conversational user interfaces, speech recognition and dialogue systems","date":"2023-07-28","arxiv_id":"2307.15493","repositories_listed":0,"syntology":null},{"url":null,"slug":"cascaded-cross-modal-transformer-for-request","title":"Cascaded Cross-Modal Transformer for Request and Complaint Detection","date":"2023-07-27","arxiv_id":"2307.15097","repositories_listed":0,"syntology":null},{"url":null,"slug":"say-goodbye-to-rnn-t-loss-a-novel-cif-based","title":"CIF-T: A Novel CIF-based Transducer Architecture for Automatic Speech Recognition","date":"2023-07-26","arxiv_id":"2307.14132","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-speaker-anonymization-of-acoustic","title":"On-Device Speaker Anonymization of Acoustic Embeddings for ASR based onFlexible Location Gradient Reversal Layer","date":"2023-07-25","arxiv_id":"2307.13343","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-punctuation-restoration-with-data","title":"Boosting Punctuation Restoration with Data Generation and Reinforcement Learning","date":"2023-07-24","arxiv_id":"2307.12949","repositories_listed":0,"syntology":null},{"url":null,"slug":"integration-of-frame-and-label-synchronous","title":"Integration of Frame- and Label-synchronous Beam Search for Streaming Encoder-decoder Speech Recognition","date":"2023-07-24","arxiv_id":"2307.12767","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-automatic-speech-recognition-via","title":"Robust Automatic Speech Recognition via WavAugment Guided Phoneme Adversarial Training","date":"2023-07-24","arxiv_id":"2307.12498","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-meta-learning-scheme-for-fast-accent-domain","title":"A meta learning scheme for fast accent domain expansion in Mandarin speech recognition","date":"2023-07-23","arxiv_id":"2307.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-integration-of-speech","title":"Exploring the Integration of Speech Separation and Recognition with Self-Supervised Learning Representation","date":"2023-07-23","arxiv_id":"2307.12231","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-confidence-aware-training-for-robust","title":"Modality Confidence Aware Training for Robust End-to-End Spoken Language Understanding","date":"2023-07-22","arxiv_id":"2307.12134","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-large-language-models-with-speech","title":"Prompting Large Language Models with Speech Recognition Abilities","date":"2023-07-21","arxiv_id":"2307.11795","repositories_listed":0,"syntology":null},{"url":null,"slug":"globally-normalising-the-transducer-for","title":"Globally Normalising the Transducer for Streaming Speech Recognition","date":"2023-07-20","arxiv_id":"2307.10975","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-pretrained-asr-and-lm-to-perform","title":"Integrating Pretrained ASR and LM to Perform Sequence Generation for Spoken Language Understanding","date":"2023-07-20","arxiv_id":"2307.11005","repositories_listed":0,"syntology":null},{"url":null,"slug":"masr-metadata-aware-speech-representation","title":"MASR: Multi-label Aware Speech Representation","date":"2023-07-20","arxiv_id":"2307.10982","repositories_listed":0,"syntology":null},{"url":null,"slug":"transsion-tsup-s-speech-recognition-system","title":"Transsion TSUP's speech recognition system for ASRU 2023 MADASR Challenge","date":"2023-07-20","arxiv_id":"2307.11778","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-visemes-for-better-visual-speech","title":"Leveraging Visemes for Better Visual Speech Representation and Lip Reading","date":"2023-07-19","arxiv_id":"2307.10157","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-adaptation-for-asr-in-low-resource","title":"Model Adaptation for ASR in low-resource Indian Languages","date":"2023-07-16","arxiv_id":"2307.07948","repositories_listed":0,"syntology":null},{"url":null,"slug":"ed-fed-a-generic-federated-learning-framework","title":"Ed-Fed: A generic federated learning framework with resource-aware client selection for edge devices","date":"2023-07-14","arxiv_id":"2307.07199","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sensitivity-of-deep-load","title":"On the Sensitivity of Deep Load Disaggregation to Adversarial Attacks","date":"2023-07-14","arxiv_id":"2307.10209","repositories_listed":0,"syntology":null},{"url":null,"slug":"replay-to-remember-continual-layer-specific","title":"Replay to Remember: Continual Layer-Specific Fine-tuning for German Speech Recognition","date":"2023-07-14","arxiv_id":"2307.07280","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-with-hidden-unit","title":"Representation Learning With Hidden Unit Clustering For Low Resource Speech Applications","date":"2023-07-14","arxiv_id":"2307.07325","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-model-size-agnostic-compute-free","title":"Towards Model-Size Agnostic, Compute-Free, Memorization-based Inference of Deep Learning","date":"2023-07-14","arxiv_id":"2307.07631","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spoken-dialect-identification-of","title":"Towards spoken dialect identification of Irish","date":"2023-07-14","arxiv_id":"2307.07436","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-integration-of-large-language","title":"Exploring the Integration of Large Language Models into Automatic Speech Recognition Systems: An Empirical Study","date":"2023-07-13","arxiv_id":"2307.06530","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-pretrained-asr-encoders-for","title":"Leveraging Pretrained ASR Encoders for Effective and Efficient End-to-End Speech Intent Classification and Slot Filling","date":"2023-07-13","arxiv_id":"2307.07057","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalization-for-bert-based-discriminative","title":"Personalization for BERT-based Discriminative Speech Recognition Rescoring","date":"2023-07-13","arxiv_id":"2307.06832","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-diarization-and-asr-with-gmm","title":"Speech Diarization and ASR with GMM","date":"2023-07-11","arxiv_id":"2307.05637","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsevsr-lightweight-and-noise-robust-visual","title":"SparseVSR: Lightweight and Noise Robust Visual Speech Recognition","date":"2023-07-10","arxiv_id":"2307.04552","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-generative-large-language-models-perform","title":"Can Generative Large Language Models Perform ASR Error Correction?","date":"2023-07-09","arxiv_id":"2307.04172","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-level-serialized-output-training-for","title":"Token-Level Serialized Output Training for Joint Streaming ASR and ST Leveraging Textual Alignments","date":"2023-07-07","arxiv_id":"2307.03354","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-hybrid-ctc-attention-end-to-end","title":"Online Hybrid CTC/Attention End-to-End Automatic Speech Recognition Architecture","date":"2023-07-05","arxiv_id":"2307.02351","repositories_listed":0,"syntology":null},{"url":null,"slug":"transgressing-the-boundaries-towards-a","title":"Transgressing the boundaries: towards a rigorous understanding of deep learning and its (non-)robustness","date":"2023-07-05","arxiv_id":"2307.02454","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-data-augmentations-and-vtln-to-reduce","title":"Using Data Augmentations and VTLN to Reduce Bias in Dutch End-to-End Speech Recognition Systems","date":"2023-07-05","arxiv_id":"2307.02009","repositories_listed":0,"syntology":null},{"url":null,"slug":"align-with-purpose-optimize-desired","title":"Align With Purpose: Optimize Desired Properties in CTC Models with a General Plug-and-Play Framework","date":"2023-07-04","arxiv_id":"2307.01715","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-norwegian-automatic-speech","title":"Boosting Norwegian Automatic Speech Recognition","date":"2023-07-04","arxiv_id":"2307.01672","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-aware-audio-grounded-generative","title":"Knowledge-Aware Audio-Grounded Generative Slot Filling for Limited Annotated Data","date":"2023-07-04","arxiv_id":"2307.01764","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribing-educational-videos-using-whisper","title":"Transcribing Educational Videos Using Whisper: A preliminary study on using AI for transcribing educational videos","date":"2023-07-04","arxiv_id":"2307.03200","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-contextual-adapters-to-improve","title":"Multilingual Contextual Adapters To Improve Custom Word Recognition In Low-resource Languages","date":"2023-07-03","arxiv_id":"2307.00759","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformer-llms-convolution-augmented-large","title":"Conformer LLMs -- Convolution Augmented Large Language Models","date":"2023-07-02","arxiv_id":"2307.00461","repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-stop-self-supervision-accent-adaptation","title":"Don't Stop Self-Supervision: Accent Adaptation of Speech Representations via Residual Adapters","date":"2023-07-02","arxiv_id":"2307.00453","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-of-non-native","title":"Automatic Speech Recognition of Non-Native Child Speech for Language Learning Applications","date":"2023-06-29","arxiv_id":"2306.16710","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-cross-utterance-context-for-asr","title":"Leveraging Cross-Utterance Context For ASR Decoding","date":"2023-06-29","arxiv_id":"2306.16903","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-transducers-through-adjacent","title":"Accelerating Transducers through Adjacent Token Merging","date":"2023-06-28","arxiv_id":"2306.16009","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-large-language-models-for-zero-shot","title":"Prompting Large Language Models for Zero-Shot Domain Adaptation in Speech Recognition","date":"2023-06-28","arxiv_id":"2306.16007","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-deep-learning-hardware","title":"A Survey on Deep Learning Hardware Accelerators for Heterogeneous HPC Platforms","date":"2023-06-27","arxiv_id":"2306.15552","repositories_listed":0,"syntology":null},{"url":null,"slug":"confidence-based-ensembles-of-end-to-end","title":"Confidence-based Ensembles of End-to-End Speech Recognition Models","date":"2023-06-27","arxiv_id":"2306.15824","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyper-parameter-adaptation-of-conformer-asr","title":"Hyper-parameter Adaptation of Conformer ASR Systems for Elderly and Dysarthric Speech Recognition","date":"2023-06-27","arxiv_id":"2306.15265","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-unsupervised-audio-pre-training","title":"Large-scale unsupervised audio pre-training for video-to-speech synthesis","date":"2023-06-27","arxiv_id":"2306.15464","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-the-gap-between-streaming-and-non","title":"Reducing the gap between streaming and non-streaming Transducer-based ASR by adaptive two-stage knowledge distillation","date":"2023-06-27","arxiv_id":"2306.15171","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-discriminative-speech","title":"Scaling Laws for Discriminative Speech Recognition Rescoring Models","date":"2023-06-27","arxiv_id":"2306.15815","repositories_listed":0,"syntology":null},{"url":null,"slug":"factorised-speaker-environment-adaptive","title":"Factorised Speaker-environment Adaptive Training of Conformer Speech Recognition Systems","date":"2023-06-26","arxiv_id":"2306.14608","repositories_listed":0,"syntology":null},{"url":null,"slug":"master-asr-achieving-multilingual-scalability","title":"Master-ASR: Achieving Multilingual Scalability and Low-Resource Adaptation in ASR with Modular Learning","date":"2023-06-23","arxiv_id":"2306.15686","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-gating-framework-for-fast-and-continuous","title":"Meta-Gating Framework for Fast and Continuous Resource Optimization in Dynamic Wireless Environments","date":"2023-06-23","arxiv_id":"2306.13277","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-chime-7-dasr-challenge-distant-meeting","title":"The CHiME-7 DASR Challenge: Distant Meeting Transcription with Multiple Devices in Diverse Scenarios","date":"2023-06-23","arxiv_id":"2306.13734","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-effective-and-compact-contextual","title":"Towards Effective and Compact Contextual Representation for Conformer Transducer Speech Recognition Systems","date":"2023-06-23","arxiv_id":"2306.13307","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiopalm-a-large-language-model-that-can","title":"AudioPaLM: A Large Language Model That Can Speak and Listen","date":"2023-06-22","arxiv_id":"2306.12925","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-role-of-audio-in-video","title":"Exploring the Role of Audio in Video Captioning","date":"2023-06-21","arxiv_id":"2306.12559","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-self-learning-with-weak-supervision","title":"Federated Self-Learning with Weak Supervision for Speech Recognition","date":"2023-06-21","arxiv_id":"2306.12015","repositories_listed":0,"syntology":null}],"record_sha256":"d19d8b4a1f00b8e6938dd49291c5dc2488bb7f51955787e93f7e85fb65a75a17","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}