{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/27","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":27,"pages_in_order":65,"rows_per_page":100,"rows":[2601,2700],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/26","next":"/task/speech-recognition/papers/28","papers":[{"url":null,"slug":"graph-meets-llm-a-novel-approach-to","title":"Graph Meets LLM: A Novel Approach to Collaborative Filtering for Robust Conversational Understanding","date":"2023-05-23","arxiv_id":"2305.14449","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-gap-in-visual-speech","title":"Improving the Gap in Visual Speech Recognition Between Normal and Silent Speech Based on Metric Learning","date":"2023-05-23","arxiv_id":"2305.14203","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-transferability-of-whisper-based","title":"On the Transferability of Whisper-based Representations for \"In-the-Wild\" Cross-Task Downstream Speech Applications","date":"2023-05-23","arxiv_id":"2305.14546","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-predictive-asr-for-latency","title":"Personalized Predictive ASR for Latency Reduction in Voice Assistants","date":"2023-05-23","arxiv_id":"2305.13794","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-speech-recognition-with-a","title":"Rethinking Speech Recognition with A Multimodal Perspective via Acoustic and Semantic Cooperative Decoding","date":"2023-05-23","arxiv_id":"2305.14049","repositories_listed":0,"syntology":null},{"url":null,"slug":"se-bridge-speech-enhancement-with-consistent","title":"SE-Bridge: Speech Enhancement with Consistent Brownian Bridge","date":"2023-05-23","arxiv_id":"2305.13796","repositories_listed":0,"syntology":null},{"url":null,"slug":"tranusr-phoneme-to-word-transcoder-based","title":"TranUSR: Phoneme-to-word Transcoder Based Unified Speech Representation Learning for Cross-lingual Speech Recognition","date":"2023-05-23","arxiv_id":"2305.13629","repositories_listed":0,"syntology":null},{"url":null,"slug":"debiased-automatic-speech-recognition-for","title":"Debiased Automatic Speech Recognition for Dysarthric Speech via Sample Reweighting with Sample Affinity Test","date":"2023-05-22","arxiv_id":"2305.13108","repositories_listed":0,"syntology":null},{"url":null,"slug":"gncformer-enhanced-self-attention-for","title":"GNCformer Enhanced Self-attention for Automatic Speech Recognition","date":"2023-05-22","arxiv_id":"2305.12755","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-domain-adaptation-for-conformer-based","title":"Modular Domain Adaptation for Conformer-Based Streaming ASR","date":"2023-05-22","arxiv_id":"2305.13408","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-generation-with-speech-synthesis-for-asr","title":"Text Generation with Speech Synthesis for ASR Data Augmentation","date":"2023-05-22","arxiv_id":"2305.16333","repositories_listed":0,"syntology":null},{"url":null,"slug":"casa-asr-context-aware-speaker-attributed-asr","title":"CASA-ASR: Context-Aware Speaker-Attributed ASR","date":"2023-05-21","arxiv_id":"2305.12459","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextualized-end-to-end-speech-recognition","title":"Contextualized End-to-End Speech Recognition with Contextual Phrase Prediction Network","date":"2023-05-21","arxiv_id":"2305.12493","repositories_listed":0,"syntology":null},{"url":null,"slug":"dccrn-kws-an-audio-bias-based-model-for-noise","title":"DCCRN-KWS: an audio bias based model for noise robust small-footprint keyword spotting","date":"2023-05-21","arxiv_id":"2305.12331","repositories_listed":0,"syntology":null},{"url":null,"slug":"hystoc-obtaining-word-confidences-for-fusion","title":"Hystoc: Obtaining word confidences for fusion of end-to-end ASR systems","date":"2023-05-21","arxiv_id":"2305.12579","repositories_listed":0,"syntology":null},{"url":"/paper/multi-head-state-space-model-for-speech","slug":"multi-head-state-space-model-for-speech","title":"Multi-Head State Space Model for Speech Recognition","date":"2023-05-21","arxiv_id":"2305.12498","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-efficacy-and-noise-robustness-of","title":"On the Efficacy and Noise-Robustness of Jointly Learned Speech Emotion and Automatic Speech Recognition","date":"2023-05-21","arxiv_id":"2305.12540","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-vad-low-latency-voice-activity","title":"Semantic VAD: Low-Latency Voice Activity Detection for Speech Interaction","date":"2023-05-21","arxiv_id":"2305.12450","repositories_listed":0,"syntology":null},{"url":null,"slug":"vakta-setu-a-speech-to-speech-machine","title":"VAKTA-SETU: A Speech-to-Speech Machine Translation Service in Select Indic Languages","date":"2023-05-21","arxiv_id":"2305.12518","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-representations-in-speech","title":"Self-supervised representations in speech-based depression detection","date":"2023-05-20","arxiv_id":"2305.12263","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-universal-phonetic-encoder-for-low","title":"Language-universal phonetic encoder for low-resource speech recognition","date":"2023-05-19","arxiv_id":"2305.11576","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-universal-phonetic-representation-in","title":"Language-Universal Phonetic Representation in Multilingual Speech Pretraining for Low-Resource Speech Recognition","date":"2023-05-19","arxiv_id":"2305.11569","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-asr-via-cross-lingual-pseudo","title":"Unsupervised ASR via Cross-Lingual Pseudo-Labeling","date":"2023-05-19","arxiv_id":"2305.13330","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lexical-aware-non-autoregressive","title":"A Lexical-aware Non-autoregressive Transformer-based ASR Model","date":"2023-05-18","arxiv_id":"2305.10839","repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-and-reliable-confidence-estimation","title":"Accurate and Reliable Confidence Estimation Based on Non-Autoregressive End-to-End Speech Recognition System","date":"2023-05-18","arxiv_id":"2305.10680","repositories_listed":0,"syntology":null},{"url":null,"slug":"ml-superb-multilingual-speech-universal","title":"ML-SUPERB: Multilingual Speech Universal PERformance Benchmark","date":"2023-05-18","arxiv_id":"2305.10615","repositories_listed":0,"syntology":null},{"url":null,"slug":"use-of-speech-impairment-severity-for","title":"Use of Speech Impairment Severity for Dysarthric Speech Recognition","date":"2023-05-18","arxiv_id":"2305.10659","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisper-kdq-a-lightweight-whisper-via-guided","title":"DQ-Whisper: Joint Distillation and Quantization for Efficient Multilingual Speech Recognition","date":"2023-05-18","arxiv_id":"2305.10788","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-local-spectro-temporal-features-for","title":"Boosting Local Spectro-Temporal Features for Speech Analysis","date":"2023-05-17","arxiv_id":"2305.10270","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-speaker-disentanglement-using","title":"Adversarial Speaker Disentanglement Using Unannotated External Data for Self-supervised Representation Based Voice Conversion","date":"2023-05-16","arxiv_id":"2305.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-agnostic-language-modeling-for-on","title":"Application-Agnostic Language Modeling for On-Device ASR","date":"2023-05-16","arxiv_id":"2305.09764","repositories_listed":0,"syntology":null},{"url":null,"slug":"critical-appraisal-of-artificial-intelligence","title":"Critical Appraisal of Artificial Intelligence-Mediated Communication","date":"2023-05-15","arxiv_id":"2305.11897","repositories_listed":0,"syntology":null},{"url":null,"slug":"ood-speech-a-large-bengali-speech-recognition","title":"OOD-Speech: A Large Bengali Speech Recognition Dataset for Out-of-Distribution Benchmarking","date":"2023-05-15","arxiv_id":"2305.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-neural-factor-analysis-for","title":"Self-supervised Neural Factor Analysis for Disentangling Utterance-level Speech Representations","date":"2023-05-14","arxiv_id":"2305.08099","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerator-aware-training-for-transducer","title":"Accelerator-Aware Training for Transducer-Based Speech Recognition","date":"2023-05-12","arxiv_id":"2305.07778","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-for-end-to-end-asr-by","title":"Continual Learning for End-to-End ASR by Averaging Domain Experts","date":"2023-05-12","arxiv_id":"2305.09681","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-sensitivity-of-automatic","title":"Investigating the Sensitivity of Automatic Speech Recognition Systems to Phonetic Variation in L2 Englishes","date":"2023-05-12","arxiv_id":"2305.07389","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-audio-text-encoders-are-effective","title":"Masked Audio Text Encoders are Effective Multi-Modal Rescorers","date":"2023-05-11","arxiv_id":"2305.07677","repositories_listed":0,"syntology":null},{"url":null,"slug":"quran-recitation-recognition-using-end-to-end","title":"Quran Recitation Recognition using End-to-End Deep Learning","date":"2023-05-10","arxiv_id":"2305.07034","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-language-dependency-for","title":"Exploration of Language Dependency for Japanese Self-Supervised Speech Representation Models","date":"2023-05-09","arxiv_id":"2305.05201","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-acoustic-and-semantic-contextual","title":"Robust Acoustic and Semantic Contextual Biasing in Neural Transducers for Speech Recognition","date":"2023-05-09","arxiv_id":"2305.05271","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-needs-decoders-efficient-estimation-of","title":"Who Needs Decoders? Efficient Estimation of Sequence-level Attributes","date":"2023-05-09","arxiv_id":"2305.05098","repositories_listed":0,"syntology":null},{"url":"/paper/fast-conformer-with-linearly-scalable","slug":"fast-conformer-with-linearly-scalable","title":"Fast Conformer with Linearly Scalable Attention for Efficient Speech Recognition","date":"2023-05-08","arxiv_id":"2305.05084","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-temporal-lip-audio-memory-for-visual","title":"Multi-Temporal Lip-Audio Memory for Visual Speech Recognition","date":"2023-05-08","arxiv_id":"2305.04542","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-steerer-novel-steering-vector","title":"Neural Steerer: Novel Steering Vector Synthesis with a Causal Neural Field over Frequency and Source Positions","date":"2023-05-08","arxiv_id":"2305.04447","repositories_listed":0,"syntology":null},{"url":null,"slug":"lookahead-when-it-matters-adaptive-non-causal","title":"Lookahead When It Matters: Adaptive Non-causal Transformers for Streaming Neural Transducers","date":"2023-05-07","arxiv_id":"2305.04159","repositories_listed":0,"syntology":null},{"url":null,"slug":"employing-hybrid-deep-neural-networks-on-dari","title":"Employing Hybrid Deep Neural Networks on Dari Speech","date":"2023-05-04","arxiv_id":"2305.03200","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-spoken-language-understanding-4","title":"End-to-end spoken language understanding using joint CTC loss and self-supervised, pretrained acoustic encoders","date":"2023-05-04","arxiv_id":"2305.02937","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-transducer-and-attention-based-encoder","title":"Hybrid Transducer and Attention based Encoder-Decoder Modeling for Speech-to-Text Tasks","date":"2023-05-04","arxiv_id":"2305.03101","repositories_listed":0,"syntology":null},{"url":null,"slug":"considerations-for-ethical-speech-recognition","title":"Considerations for Ethical Speech Recognition Datasets","date":"2023-05-03","arxiv_id":"2305.02081","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-integration-of-pipeline-and","title":"A Study on the Integration of Pipeline and E2E SLU systems for Spoken Semantic Parsing toward STOP Quality Challenge","date":"2023-05-02","arxiv_id":"2305.01620","repositories_listed":0,"syntology":null},{"url":null,"slug":"lessons-learned-in-atco2-5000-hours-of-air","title":"Lessons Learned in ATCO2: 5000 hours of Air Traffic Control Communications for Robust Automatic Speech Recognition and Understanding","date":"2023-05-02","arxiv_id":"2305.01155","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-learning-techniques-for-3","title":"A Review of Deep Learning Techniques for Speech Processing","date":"2023-04-30","arxiv_id":"2305.00359","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-non-native-speech-corpus-featuring","title":"Building a Non-native Speech Corpus Featuring Chinese-English Bilingual Children: Compilation and Rationale","date":"2023-04-30","arxiv_id":"2305.00446","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-spatio-temporal-facial","title":"Deep Learning-based Spatio Temporal Facial Feature Visual Speech Recognition","date":"2023-04-30","arxiv_id":"2305.00552","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-transfer-learning-for-automatic-speech","title":"Deep Transfer Learning for Automatic Speech Recognition: Towards Better Generalization","date":"2023-04-27","arxiv_id":"2304.14535","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-shared-speech-text","title":"Understanding Shared Speech-Text Representations","date":"2023-04-27","arxiv_id":"2304.14514","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-spoken-information-queries-for","title":"Modeling Spoken Information Queries for Virtual Assistants: Open Problems, Challenges and Opportunities","date":"2023-04-25","arxiv_id":"2304.13149","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-regularised-minimum-latency-training-for","title":"Self-regularised Minimum Latency Training for Streaming Transformer-based Speech Recognition","date":"2023-04-24","arxiv_id":"2304.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-neural-networks-and-long-short-term","title":"Recurrent Neural Networks and Long Short-Term Memory Networks: Tutorial and Survey","date":"2023-04-22","arxiv_id":"2304.11461","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-end-to-end-approaches-for","title":"Non-autoregressive End-to-end Approaches for Joint Automatic Speech Recognition and Spoken Language Understanding","date":"2023-04-21","arxiv_id":"2304.10869","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-universal-defense-for-query-based","title":"Towards the Universal Defense for Query-Based Audio Adversarial Attacks","date":"2023-04-20","arxiv_id":"2304.10088","repositories_listed":0,"syntology":null},{"url":null,"slug":"security-and-privacy-problems-in-voice","title":"Security and Privacy Problems in Voice Assistant Applications: A Survey","date":"2023-04-19","arxiv_id":"2304.09486","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-nearest-neighbour-phrase-mining","title":"Approximate Nearest Neighbour Phrase Mining for Contextual Speech Recognition","date":"2023-04-18","arxiv_id":"2304.08862","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-chuck-convolution-for-unified","title":"Dynamic Chunk Convolution for Unified Streaming and Non-Streaming Conformer ASR","date":"2023-04-18","arxiv_id":"2304.09325","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-transferable-audio-adversarial","title":"Towards the Transferable Audio Adversarial Attack via Ensemble Methods","date":"2023-04-18","arxiv_id":"2304.08811","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-short-video-rumor-detection-system","title":"Multimodal Short Video Rumor Detection System Based on Contrastive Learning","date":"2023-04-17","arxiv_id":"2304.08401","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-virtual-simulation-pilot-agent-for-training","title":"A Virtual Simulation-Pilot Agent for Training of Air Traffic Controllers","date":"2023-04-16","arxiv_id":"2304.07842","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-ctc-alignment-based-non-autoregressive","title":"A CTC Alignment-based Non-autoregressive Transformer for End-to-end Automatic Speech Recognition","date":"2023-04-15","arxiv_id":"2304.07611","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-speaker-anonymization-on","title":"Evaluation of Speaker Anonymization on Emotional Speech","date":"2023-04-15","arxiv_id":"2305.01759","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-document-grounded-dialog","title":"Task-oriented Document-Grounded Dialog Systems by HLTPR@RWTH for DSTC9 and DSTC10","date":"2023-04-14","arxiv_id":"2304.07101","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-tensor-low-cycle-rank-approximation","title":"Solving Tensor Low Cycle Rank Approximation","date":"2023-04-13","arxiv_id":"2304.06594","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-contrastive-predictive-coding","title":"Regularizing Contrastive Predictive Coding for Speech Applications","date":"2023-04-12","arxiv_id":"2304.05974","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-reconstruction-from-silent-tongue-and","title":"Speech Reconstruction from Silent Tongue and Lip Articulation By Pseudo Target Generation and Domain Adversarial Training","date":"2023-04-12","arxiv_id":"2304.05574","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-t-simplify-the-transformer-network-by","title":"Sim-T: Simplify the Transformer Network by Multiplexing Technique for Speech Recognition","date":"2023-04-11","arxiv_id":"2304.04991","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2code-restore-clean-speech-representations","title":"Wav2code: Restore Clean Speech Representations via Codebook Lookup for Noise-Robust ASR","date":"2023-04-11","arxiv_id":"2304.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-feature-fusion-enhancing","title":"Adaptive Feature Fusion: Enhancing Generalization in Deep Learning Models","date":"2023-04-04","arxiv_id":"2304.03290","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-and-accurate-self-supervised","title":"Scalable and Accurate Self-supervised Multimodal Representation Learning without Aligned Video and Text Data","date":"2023-04-04","arxiv_id":"2304.02080","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-attention-neural-transducers-for","title":"Dual-Attention Neural Transducers for Efficient Wake Word Spotting in Speech Recognition","date":"2023-04-03","arxiv_id":"2304.01905","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-based-source","title":"Self-Supervised Learning-Based Source Separation for Meeting Data","date":"2023-04-03","arxiv_id":"2304.00871","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-word-error-rate-estimation-e","title":"Multilingual Word Error Rate Estimation: e-WER3","date":"2023-04-02","arxiv_id":"2304.00649","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialog-act-guided-contextual-adapter-for","title":"Dialog act guided contextual adapter for personalized speech recognition","date":"2023-03-31","arxiv_id":"2303.17799","repositories_listed":0,"syntology":null},{"url":"/paper/improving-the-previous-state-of-the-art","slug":"improving-the-previous-state-of-the-art","title":"Improving the previous state-of-the-art Frisian ASR by fine-tuning XLS-R","date":"2023-03-31","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lego-features-exporting-modular-encoder","title":"Lego-Features: Exporting modular encoder features for streaming and deliberation ASR","date":"2023-03-31","arxiv_id":"2304.00173","repositories_listed":0,"syntology":null},{"url":"/paper/the-edinburgh-international-accents-of","slug":"the-edinburgh-international-accents-of","title":"The Edinburgh International Accents of English Corpus: Towards the Democratization of English ASR","date":"2023-03-31","arxiv_id":"2303.18110","repositories_listed":0,"syntology":null},{"url":null,"slug":"procter-pronunciation-aware-contextual","title":"PROCTER: PROnunciation-aware ConTextual adaptER for personalized speech recognition in neural transducers","date":"2023-03-30","arxiv_id":"2303.17131","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthvsr-scaling-up-visual-speech-recognition","title":"SynthVSR: Scaling Up Visual Speech Recognition With Synthetic Supervision","date":"2023-03-30","arxiv_id":"2303.17200","repositories_listed":0,"syntology":null},{"url":null,"slug":"avformer-injecting-vision-into-frozen-speech","title":"AVFormer: Injecting Vision into Frozen Speech Models for Zero-Shot AV-ASR","date":"2023-03-29","arxiv_id":"2303.16501","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-unsupervised-and-supervised-learning","title":"Joint unsupervised and supervised learning for context-aware language identification","date":"2023-03-29","arxiv_id":"2303.16511","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-is-all-you-need-personalizing-asr-models","title":"Text is All You Need: Personalizing ASR Models using Controllable Speech Synthesis","date":"2023-03-27","arxiv_id":"2303.14885","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deliberation-based-joint-acoustic-and-text","title":"A Deliberation-based Joint Acoustic and Text Decoder","date":"2023-03-23","arxiv_id":"2303.15293","repositories_listed":0,"syntology":null},{"url":"/paper/beyond-universal-transformer-block-reusing","slug":"beyond-universal-transformer-block-reusing","title":"Beyond Universal Transformer: block reusing with adaptor in Transformer for automatic speech recognition","date":"2023-03-23","arxiv_id":"2303.13072","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-unsupervised-speech-recognition","title":"Enhancing Unsupervised Speech Recognition with Diffusion GANs","date":"2023-03-23","arxiv_id":"2303.13559","repositories_listed":0,"syntology":null},{"url":null,"slug":"msat-biologically-inspired-multi-stage","title":"MSAT: Biologically Inspired Multi-Stage Adaptive Threshold for Conversion of Spiking Neural Networks","date":"2023-03-23","arxiv_id":"2303.13080","repositories_listed":0,"syntology":null},{"url":null,"slug":"pyramid-multi-branch-fusion-dcnn-with-multi","title":"Pyramid Multi-branch Fusion DCNN with Multi-Head Self-Attention for Mandarin Speech Recognition","date":"2023-03-23","arxiv_id":"2303.13243","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-turkish-speech-recognition-via","title":"Exploring Turkish Speech Recognition via Hybrid CTC/Attention Architecture and Multi-feature Fusion Network","date":"2023-03-22","arxiv_id":"2303.12300","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-with-speech","title":"Self-supervised Learning with Speech Modulation Dropout","date":"2023-03-22","arxiv_id":"2303.12908","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-integration-of-speech-separation","title":"End-to-End Integration of Speech Separation and Voice Activity Detection for Low-Latency Diarization of Telephone Conversations","date":"2023-03-21","arxiv_id":"2303.12002","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-in-speech-processing-a-survey","title":"Transformers in Speech Processing: A Survey","date":"2023-03-21","arxiv_id":"2303.11607","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switching-text-generation-and-injection","title":"Code-Switching Text Generation and Injection in Mandarin-English ASR","date":"2023-03-20","arxiv_id":"2303.10949","repositories_listed":0,"syntology":null}],"record_sha256":"48402d1b1552e54aaefb43ce4ca38ab7b77279b1f0ad04a6b0a37e8cffedb823","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}