{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/29","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":29,"pages_in_order":58,"rows_per_page":100,"rows":[2801,2900],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/28","next":"/task/speech-recognition-1/papers/30","papers":[{"url":null,"slug":"explicit-intensity-control-for-accented-text","title":"Explicit Intensity Control for Accented Text-to-speech","date":"2022-10-27","arxiv_id":"2210.15364","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-effective-distillation-of-self","title":"Exploring Effective Distillation of Self-Supervised Speech Models for Automatic Speech Recognition","date":"2022-10-27","arxiv_id":"2210.15631","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-more-of-your-data-minimal-effort-data","title":"Make More of Your Data: Minimal Effort Data Augmentation for Automatic Speech Recognition and Translation","date":"2022-10-27","arxiv_id":"2210.15398","repositories_listed":0,"syntology":null},{"url":null,"slug":"san-a-robust-end-to-end-asr-model","title":"SAN: a robust end-to-end ASR model architecture","date":"2022-10-27","arxiv_id":"2210.15285","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulating-realistic-speech-overlaps-improves","title":"Simulating realistic speech overlaps improves multi-talker ASR","date":"2022-10-27","arxiv_id":"2210.15715","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-voice-conversion-via-intermediate","title":"Streaming Voice Conversion Via Intermediate Bottleneck Features And Non-streaming Teacher Guidance","date":"2022-10-27","arxiv_id":"2210.15158","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-autoregressive-speech-recognition","title":"Training Autoregressive Speech Recognition Models with Limited in-domain Supervision","date":"2022-10-27","arxiv_id":"2210.15135","repositories_listed":0,"syntology":null},{"url":null,"slug":"trscore-a-novel-gpt-based-readability-scorer","title":"TRScore: A Novel GPT-based Readability Scorer for ASR Segmentation and Punctuation model evaluation and selection","date":"2022-10-27","arxiv_id":"2210.15104","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-cloak-intelligibility-naturalness-timbre","title":"V-Cloak: Intelligibility-, Naturalness- & Timbre-Preserving Real-Time Voice Anonymization","date":"2022-10-27","arxiv_id":"2210.15140","repositories_listed":0,"syntology":null},{"url":null,"slug":"weight-averaging-a-simple-yet-effective","title":"Weight Averaging: A Simple Yet Effective Method to Overcome Catastrophic Forgetting in Automatic Speech Recognition","date":"2022-10-27","arxiv_id":"2210.15282","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-use-of-large-pre-trained-models-for","title":"Efficient Utilization of Large Pre-Trained Models for Low Resource ASR","date":"2022-10-26","arxiv_id":"2210.15445","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-to-intent-prediction-to","title":"End-to-End Speech to Intent Prediction to improve E-commerce Customer Support Voicebot in Hindi and English","date":"2022-10-26","arxiv_id":"2211.07710","repositories_listed":0,"syntology":null},{"url":null,"slug":"four-in-one-a-joint-approach-to-inverse-text","title":"Four-in-One: A Joint Approach to Inverse Text Normalization, Punctuation, Capitalization, and Disfluency for Automatic Speech Recognition","date":"2022-10-26","arxiv_id":"2210.15063","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-to-speech-translation","title":"Improving Speech-to-Speech Translation Through Unlabeled Text","date":"2022-10-26","arxiv_id":"2210.14514","repositories_listed":0,"syntology":null},{"url":null,"slug":"pronunciation-generation-for-foreign-language","title":"Pronunciation Generation for Foreign Language Words in Intra-Sentential Code-Switching Speech Recognition","date":"2022-10-26","arxiv_id":"2210.14691","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-speech-segmentation-using-acousto","title":"Smart Speech Segmentation using Acousto-Linguistic Features with look-ahead","date":"2022-10-26","arxiv_id":"2210.14446","repositories_listed":0,"syntology":null},{"url":null,"slug":"ufo2-a-unified-pre-training-framework-for","title":"UFO2: A unified pre-training framework for online and offline speech recognition","date":"2022-10-26","arxiv_id":"2210.14515","repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistic-enhanced-transformer-with-ctc","title":"Linguistic-Enhanced Transformer with CTC Embedding for Speech Recognition","date":"2022-10-25","arxiv_id":"2210.14725","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-effect-of-domain-selection","title":"Investigating self-supervised, weakly supervised and fully supervised training approaches for multi-domain automatic speech recognition: a study on Bangladeshi Bangla","date":"2022-10-24","arxiv_id":"2210.12921","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-domain-speech-enhancement-for-robust","title":"Time-Domain Speech Enhancement for Robust Automatic Speech Recognition","date":"2022-10-24","arxiv_id":"2210.13318","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-contrastive-self-supervised-pre","title":"Guided contrastive self-supervised pre-training for automatic speech recognition","date":"2022-10-22","arxiv_id":"2210.12335","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-visual-context-improve-automatic-speech","title":"Can Visual Context Improve Automatic Speech Recognition for an Embodied Agent?","date":"2022-10-21","arxiv_id":"2210.13189","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-lstm-spoken-term-detection-using-wav2vec","title":"Deep LSTM Spoken Term Detection using Wav2Vec 2.0 Recognizer","date":"2022-10-21","arxiv_id":"2210.11885","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-bilingual-neural-transducer-with","title":"Optimizing Bilingual Neural Transducer with Synthetic Code-switching Text Generation","date":"2022-10-21","arxiv_id":"2210.12214","repositories_listed":0,"syntology":null},{"url":null,"slug":"anchored-speech-recognition-with-neural","title":"Anchored Speech Recognition with Neural Transducers","date":"2022-10-20","arxiv_id":"2210.11588","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-semi-supervised-end-to-end","title":"Improving Semi-supervised End-to-end Automatic Speech Recognition using CycleGAN and Inter-domain Losses","date":"2022-10-20","arxiv_id":"2210.11642","repositories_listed":0,"syntology":null},{"url":"/paper/g-augment-searching-for-the-meta-structure-of","slug":"g-augment-searching-for-the-meta-structure-of","title":"G-Augment: Searching for the Meta-Structure of Data Augmentation Policies for ASR","date":"2022-10-19","arxiv_id":"2210.10879","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-and-age-invariant-training-for-child","title":"Speaker- and Age-Invariant Training for Child Acoustic Modeling Using Adversarial Multi-Task Learning","date":"2022-10-19","arxiv_id":"2210.10231","repositories_listed":0,"syntology":null},{"url":null,"slug":"tourist-guidance-robot-based-on-hyperclova","title":"Tourist Guidance Robot Based on HyperCLOVA","date":"2022-10-19","arxiv_id":"2210.10400","repositories_listed":0,"syntology":null},{"url":null,"slug":"it-s-a-long-way-layer-wise-relevance","title":"Layer-wise Relevance Propagation for Echo State Networks applied to Earth System Variability","date":"2022-10-18","arxiv_id":"2210.09958","repositories_listed":0,"syntology":null},{"url":null,"slug":"maestro-u-leveraging-joint-speech-text","title":"Maestro-U: Leveraging joint speech-text representation learning for zero supervised speech ASR","date":"2022-10-18","arxiv_id":"2210.10027","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalization-of-ctc-speech-recognition","title":"Towards Personalization of CTC Speech Recognition Models with Contextual Adapters and Adaptive Boosting","date":"2022-10-18","arxiv_id":"2210.09510","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-and-effective-unsupervised-speech-1","title":"Simple and Effective Unsupervised Speech Translation","date":"2022-10-18","arxiv_id":"2210.10191","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-treatise-on-fst-lattice-based-mmi-training","title":"A Treatise On FST Lattice Based MMI Training","date":"2022-10-17","arxiv_id":"2210.08918","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-pseudo-labeling-from-the-start","title":"Continuous Pseudo-Labeling from the Start","date":"2022-10-17","arxiv_id":"2210.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-agnostic-code-switching-in-end-to","title":"Language-agnostic Code-Switching in Sequence-To-Sequence Speech Recognition","date":"2022-10-17","arxiv_id":"2210.08992","repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-8-bit-quantization-for-on-device-speech","title":"Sub-8-bit quantization for on-device speech recognition: a regularization-free approach","date":"2022-10-17","arxiv_id":"2210.09188","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-aware-non-autoregressive-spell","title":"Acoustic-aware Non-autoregressive Spell Correction with Mask Sample Decoding","date":"2022-10-16","arxiv_id":"2210.08665","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-invariant-representation-and-risk","title":"Learning Invariant Representation and Risk Minimized for Unsupervised Accent Domain Adaptation","date":"2022-10-15","arxiv_id":"2210.08182","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-jointly-transcribe-and-subtitle","title":"Learning to Jointly Transcribe and Subtitle for End-to-End Spontaneous Speech Recognition","date":"2022-10-14","arxiv_id":"2210.07771","repositories_listed":0,"syntology":null},{"url":null,"slug":"levoice-asr-systems-for-the-iscslp-2022","title":"LeVoice ASR Systems for the ISCSLP 2022 Intelligent Cockpit Speech Recognition Challenge","date":"2022-10-14","arxiv_id":"2210.07749","repositories_listed":0,"syntology":null},{"url":null,"slug":"hubert-tr-reviving-turkish-automatic-speech","title":"Experiments on Turkish ASR with Self-Supervised Speech Representation Learning","date":"2022-10-13","arxiv_id":"2210.07323","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-zero-resource-speech-recognition","title":"Multilingual Zero Resource Speech Recognition Base on Self-Supervise Pre-Trained Acoustic Models","date":"2022-10-13","arxiv_id":"2210.06936","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-ensemble-teacher-student-learning-approach","title":"An Ensemble Teacher-Student Learning Approach with Poisson Sub-sampling to Differential Privacy Preserving Speech Recognition","date":"2022-10-12","arxiv_id":"2210.06382","repositories_listed":0,"syntology":null},{"url":null,"slug":"summary-on-the-iscslp-2022-chinese-english","title":"Summary on the ISCSLP 2022 Chinese-English Code-Switching ASR Challenge","date":"2022-10-12","arxiv_id":"2210.06091","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-study-on-private-aggregation","title":"An Experimental Study on Private Aggregation of Teacher Ensemble Learning for End-to-End Speech Recognition","date":"2022-10-11","arxiv_id":"2210.05614","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-of-low-resource","title":"Automatic Speech Recognition of Low-Resource Languages Based on Chukchi","date":"2022-10-11","arxiv_id":"2210.05726","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-soft-and-hard-target-rnn-t","title":"Comparison of Soft and Hard Target RNN-T Distillation for Large-scale ASR","date":"2022-10-11","arxiv_id":"2210.05793","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-alignments-improve-autoregressive","title":"CTC Alignments Improve Autoregressive Translation","date":"2022-10-11","arxiv_id":"2210.05200","repositories_listed":0,"syntology":null},{"url":null,"slug":"inner-speech-recognition-through","title":"Inner speech recognition through electroencephalographic signals","date":"2022-10-11","arxiv_id":"2210.06472","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-deliberation-for-multilingual-asr","title":"Scaling Up Deliberation for Multilingual ASR","date":"2022-10-11","arxiv_id":"2210.05785","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-punctuation-for-long-form-dictation","title":"Streaming Punctuation for Long-form Dictation with Transformers","date":"2022-10-11","arxiv_id":"2210.05756","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloud-based-automatic-speech-recognition","title":"Cloud-based Automatic Speech Recognition Systems for Southeast Asian Languages","date":"2022-10-07","arxiv_id":"2210.03580","repositories_listed":0,"syntology":null},{"url":null,"slug":"pronunciation-modeling-of-foreign-words-for","title":"Pronunciation Modeling of Foreign Words for Mandarin ASR by Considering the Effect of Language Transfer","date":"2022-10-07","arxiv_id":"2210.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"damage-control-during-domain-adaptation-for","title":"Damage Control During Domain Adaptation for Transducer Based Automatic Speech Recognition","date":"2022-10-06","arxiv_id":"2210.03255","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-dataset-generation-for-privacy","title":"Synthetic Dataset Generation for Privacy-Preserving Machine Learning","date":"2022-10-06","arxiv_id":"2210.03205","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switching-without-switching-language","title":"Code-Switching without Switching: Language Agnostic End-to-End Speech Translation","date":"2022-10-04","arxiv_id":"2210.01512","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-acoustic-feature-transformation-in","title":"Efficient acoustic feature transformation in mismatched environments using a Guided-GAN","date":"2022-10-03","arxiv_id":"2210.00721","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-limited-samples-meta-learning","title":"Learning with Limited Samples -- Meta-Learning and Applications to Communication Systems","date":"2022-10-03","arxiv_id":"2210.02515","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-transformer-convolutional-and","title":"A Comparison of Transformer, Convolutional, and Recurrent Neural Networks on Phoneme Recognition","date":"2022-10-01","arxiv_id":"2210.00367","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-train-a-language-model-inside-an-end","title":"Can We Train a Language Model Inside an End-to-End ASR Model? - Investigating Effective Implicit Language Modeling","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fashioning-local-designs-from-generic-speech","title":"Fashioning Local Designs from Generic Speech Technologies in an Australian Aboriginal Community","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-code-switched-asr-with-linguistic","title":"Improving Code-switched ASR with Linguistic Information","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-impact-of-asr-errors-on","title":"Investigating the Impact of ASR Errors on Spoken Implicit Discourse Relation Recognition","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"keyphrase-prediction-from-video-transcripts","title":"Keyphrase Prediction from Video Transcripts: New Dataset and Directions","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"language-specific-effects-on-automatic-speech","title":"Language-specific Effects on Automatic Speech Recognition Errors for World Englishes","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mian-xiang-transformer-mo-xing-de-meng-gu-yu","title":"面向 Transformer 模型的蒙古语语音识别词特征编码方法(Researching of the Mongolian word encoding method based on Transformer Mongolian speech recognition)","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-progressive-compression-of","title":"Multi-stage Progressive Compression of Conformer Transducer for On-device Speech Recognition","date":"2022-10-01","arxiv_id":"2210.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"rong-he-wai-bu-yu-yan-zhi-shi-de-liu-shi-yue","title":"融合外部语言知识的流式越南语语音识别(Streaming Vietnamese Speech Recognition Based on Fusing External Vietnamese Language Knowledge)","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-disfluency-detection-for-indian","title":"Zero-shot Disfluency Detection for Indian Languages","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-sparse-and-monotonic-attention-for","title":"Adaptive Sparse and Monotonic Attention for Transformer-based Automatic Speech Recognition","date":"2022-09-30","arxiv_id":"2209.15176","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-signal-dereverberation-for-machine","title":"Blind Signal Dereverberation for Machine Speech Recognition","date":"2022-09-30","arxiv_id":"2210.00117","repositories_listed":0,"syntology":null},{"url":null,"slug":"convrnn-t-convolutional-augmented-recurrent","title":"ConvRNN-T: Convolutional Augmented Recurrent Neural Network Transducers for Streaming Speech Recognition","date":"2022-09-29","arxiv_id":"2209.14868","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-effective-performant-named-entity","title":"An Effective, Performant Named Entity Recognition System for Noisy Business Telephone Conversation Transcripts","date":"2022-09-27","arxiv_id":"2209.13736","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-for-speech-1","title":"Unsupervised domain adaptation for speech recognition with unsupervised error correction","date":"2022-09-24","arxiv_id":"2209.12043","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-asr-model-quality-on-disordered","title":"Assessing ASR Model Quality on Disordered Speech using BERTScore","date":"2022-09-21","arxiv_id":"2209.10591","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-conformers-via-sharing","title":"Parameter-Efficient Conformers via Sharing Sparsely-Gated Experts for End-to-End Speech Recognition","date":"2022-09-17","arxiv_id":"2209.08326","repositories_listed":0,"syntology":null},{"url":null,"slug":"distribution-aware-metrics-for-conditional","title":"Distribution Aware Metrics for Conditional Natural Language Generation","date":"2022-09-15","arxiv_id":"2209.07518","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-parallel-voice-conversion-for-asr","title":"Non-Parallel Voice Conversion for ASR Augmentation","date":"2022-09-15","arxiv_id":"2209.06987","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-universally-deployable-asr-frontend-for","title":"A Universally-Deployable ASR Frontend for Joint Acoustic Echo Cancellation, Speech Enhancement, and Voice Separation","date":"2022-09-14","arxiv_id":"2209.06410","repositories_listed":0,"syntology":null},{"url":null,"slug":"essumm-extractive-speech-summarization-from","title":"ESSumm: Extractive Speech Summarization from Untranscribed Meeting","date":"2022-09-14","arxiv_id":"2209.06913","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-pruning-improving-neural-network","title":"Federated Pruning: Improving Neural Network Efficiency with Federated Learning","date":"2022-09-14","arxiv_id":"2209.06359","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-self-attention-head-diversity-for","title":"Analysis of Self-Attention Head Diversity for Conformer-based Automatic Speech Recognition","date":"2022-09-13","arxiv_id":"2209.06096","repositories_listed":0,"syntology":null},{"url":null,"slug":"bangla-wave-improving-bangla-automatic-speech","title":"Bangla-Wave: Improving Bangla Automatic Speech Recognition Utilizing N-gram Language Models","date":"2022-09-13","arxiv_id":"2209.12650","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-asr-pathways-a-sparse-multilingual","title":"Learning ASR pathways: A sparse multilingual ASR model","date":"2022-09-13","arxiv_id":"2209.05735","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-end-to-end-multilingual-speech","title":"Streaming End-to-End Multilingual Speech Recognition with Joint Language Identification","date":"2022-09-13","arxiv_id":"2209.06058","repositories_listed":0,"syntology":null},{"url":null,"slug":"vararray-meets-t-sot-advancing-the-state-of","title":"VarArray Meets t-SOT: Advancing the State of the Art of Streaming Distant Conversational Speech Recognition","date":"2022-09-12","arxiv_id":"2209.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-wav2vec2-for-speech-recognition-on","title":"Applying wav2vec2 for Speech Recognition on Bengali Common Voices Dataset","date":"2022-09-11","arxiv_id":"2209.06581","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexicon-and-attention-based-handwritten-text","title":"Lexicon and Attention based Handwritten Text Recognition System","date":"2022-09-11","arxiv_id":"2209.04817","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversion-of-acoustic-signal-speech-into","title":"Conversion of Acoustic Signal (Speech) Into Text By Digital Filter using Natural Language Processing","date":"2022-09-09","arxiv_id":"2209.04189","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-end-to-end-target-speaker-asr","title":"Streaming Target-Speaker ASR with Neural Transducer","date":"2022-09-09","arxiv_id":"2209.04175","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-transformer-language-model-for","title":"Multilingual Transformer Language Model for Speech Recognition in Low-resource Languages","date":"2022-09-08","arxiv_id":"2209.04041","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-dependent-structure-for-utterances","title":"Modeling Dependent Structure for Utterances in ASR Evaluation","date":"2022-09-07","arxiv_id":"2209.05281","repositories_listed":0,"syntology":null},{"url":null,"slug":"plant-species-classification-using-transfer","title":"Plant Species Classification Using Transfer Learning by Pretrained Classifier VGG-19","date":"2022-09-07","arxiv_id":"2209.03076","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-the-knowledge-of-bert-for-ctc","title":"Distilling the Knowledge of BERT for CTC-based ASR","date":"2022-09-05","arxiv_id":"2209.02030","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-deep-learning-aided-wireless-channel","title":"Towards Deep Learning-aided Wireless Channel Estimation and Channel State Information Feedback for 6G","date":"2022-09-05","arxiv_id":"2209.01724","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-sparse-expert-models-in-deep","title":"A Review of Sparse Expert Models in Deep Learning","date":"2022-09-04","arxiv_id":"2209.01667","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-fourier-attack-for-time-series","title":"Universal Fourier Attack for Time Series","date":"2022-09-02","arxiv_id":"2209.00757","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-wavelet-transform-based-scheme-to-extract","title":"A Wavelet Transform Based Scheme to Extract Speech Pitch and Formant Frequencies","date":"2022-09-01","arxiv_id":"2209.00733","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-enhanced-citrinet-for-speech","title":"Attention Enhanced Citrinet for Speech Recognition","date":"2022-09-01","arxiv_id":"2209.00261","repositories_listed":0,"syntology":null}],"record_sha256":"456972db63a10e1bcd3afb2ae62443e92a7fdf2c5b09eba618527c4e849b8d82","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}