{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/18","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":31,"rows_per_page":100,"rows":[1701,1800],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/17","next":"/task/automatic-speech-recognition/papers/19","papers":[{"url":null,"slug":"polyphonic-pitch-detection-with-convolutional","title":"Polyphonic pitch detection with convolutional recurrent neural networks","date":"2022-02-04","arxiv_id":"2202.02115","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-cuhk-tencent-speaker-diarization-system","title":"The CUHK-TENCENT speaker diarization system for the ICASSP 2022 multi-channel multi-party meeting transcription challenge","date":"2022-02-04","arxiv_id":"2202.01986","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-speech-recognition-and-audio-captioning","title":"Joint Speech Recognition and Audio Captioning","date":"2022-02-03","arxiv_id":"2202.01405","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-royalflush-system-of-speech-recognition","title":"The RoyalFlush System of Speech Recognition for M2MeT Challenge","date":"2022-02-03","arxiv_id":"2202.01614","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-aware-end-to-end-neural-diarization","title":"ASR-Aware End-to-end Neural Diarization","date":"2022-02-02","arxiv_id":"2202.01286","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-correction-in-asr-using-sequence-to","title":"Error Correction in ASR using Sequence-to-Sequence Models","date":"2022-02-02","arxiv_id":"2202.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"rescorebert-discriminative-speech-recognition","title":"RescoreBERT: Discriminative Speech Recognition Rescoring with BERT","date":"2022-02-02","arxiv_id":"2202.01094","repositories_listed":0,"syntology":null},{"url":null,"slug":"bea-base-a-benchmark-for-asr-of-spontaneous","title":"BEA-Base: A Benchmark for ASR of Spontaneous Hungarian","date":"2022-02-01","arxiv_id":"2202.00601","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-dependencies-in-adversarial-attacks","title":"Language Dependencies in Adversarial Attacks on Speech Recognition Systems","date":"2022-02-01","arxiv_id":"2202.00399","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-automatic-speech-recognition","title":"Visualizing Automatic Speech Recognition -- Means for a Better Understanding?","date":"2022-02-01","arxiv_id":"2202.00673","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-language-context-confusion-for-end","title":"Reducing language context confusion for end-to-end code-switching automatic speech recognition","date":"2022-01-28","arxiv_id":"2201.12155","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiment-aware-automatic-speech-recognition","title":"Sentiment-Aware Automatic Speech Recognition pre-training for enhanced Speech Emotion Recognition","date":"2022-01-27","arxiv_id":"2201.11826","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-dysarthric-speech-using-multi","title":"Synthesizing Dysarthric Speech Using Multi-talker TTS for Dysarthric Speech Recognition","date":"2022-01-27","arxiv_id":"2201.11571","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-decoder-transformer-for-end-to-end","title":"On the Effectiveness of Pinyin-Character Dual-Decoding for End-to-End Mandarin Chinese ASR","date":"2022-01-26","arxiv_id":"2201.10792","repositories_listed":0,"syntology":null},{"url":"/paper/the-norwegian-parliamentary-speech-corpus","slug":"the-norwegian-parliamentary-speech-corpus","title":"The Norwegian Parliamentary Speech Corpus","date":"2022-01-26","arxiv_id":"2201.10881","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-non-autoregressive-end-to-end","title":"Improving non-autoregressive end-to-end speech recognition with pre-trained acoustic and language models","date":"2022-01-25","arxiv_id":"2201.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-fusion-of-acoustic-and-text","title":"Improving the fusion of acoustic and text representations in RNN-T","date":"2022-01-25","arxiv_id":"2201.10240","repositories_listed":0,"syntology":null},{"url":null,"slug":"run-and-back-stitch-search-novel-block","title":"Run-and-back stitch search: novel block synchronous decoding for streaming encoder-decoder ASR","date":"2022-01-25","arxiv_id":"2201.10190","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-video-front-ends-for-audio","title":"Transformer-Based Video Front-Ends for Audio-Visual Speech Recognition for Single and Multi-Person Video","date":"2022-01-25","arxiv_id":"2201.10439","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-noise-robust-self-supervised-pre-training","title":"A Noise-Robust Self-supervised Pre-training Model Based Speech Representation Learning for Automatic Speech Recognition","date":"2022-01-22","arxiv_id":"2201.08930","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-bad-are-artifacts-analyzing-the-impact-of","title":"How Bad Are Artifacts?: Analyzing the Impact of Speech Enhancement Errors on ASR","date":"2022-01-18","arxiv_id":"2201.06685","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-and-automatic-speech-recognition","title":"Human and Automatic Speech Recognition Performance on German Oral History Interviews","date":"2022-01-18","arxiv_id":"2201.06841","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-textless-spoken-question-answering-with","title":"DUAL: Textless Spoken Question Answering with Speech Discrete Unit Adaptive Learning","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"red-ace-robust-error-detection-for-asr-using","title":"RED-ACE: Robust Error Detection for ASR using Confidence Embeddings","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-progress-in-the-cuhk-dysarthric-speech","title":"Recent Progress in the CUHK Dysarthric Speech Recognition System","date":"2022-01-15","arxiv_id":"2201.05845","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-enhance-or-not-neural-network","title":"Learning to Enhance or Not: Neural Network-Based Switching of Enhanced and Observed Signals for Overlapping Speech Recognition","date":"2022-01-11","arxiv_id":"2201.03881","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-likelihood-ratio-based-domain-adaptation","title":"A Likelihood Ratio based Domain Adaptation Method for E2E Models","date":"2022-01-10","arxiv_id":"2201.03655","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-asr-post-processing-system-for","title":"Cross-Modal ASR Post-Processing System for Error Correction and Utterance Rejection","date":"2022-01-10","arxiv_id":"2201.03313","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-sql-towards-speech-driven-sql-query","title":"Speech-to-SQL: Towards Speech-driven SQL Query Generation From Natural Language Question","date":"2022-01-04","arxiv_id":"2201.01209","repositories_listed":0,"syntology":null},{"url":null,"slug":"tencent-mvse-a-large-scale-benchmark-dataset","title":"Tencent-MVSE: A Large-Scale Benchmark Dataset for Multi-Modal Video Similarity Evaluation","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-dialect-arabic-speech-recognition","title":"Multi-Dialect Arabic Speech Recognition","date":"2021-12-25","arxiv_id":"2112.14678","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-based-consistency","title":"Multi-Variant Consistency based Self-supervised Learning for Robust Automatic Speech Recognition","date":"2021-12-23","arxiv_id":"2112.12522","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-quality-and-pitch-features-in","title":"Voice Quality and Pitch Features in Transformer-Based Speech Recognition","date":"2021-12-21","arxiv_id":"2112.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-knowledge-in-end-to-end-automatic","title":"Integrating Knowledge in End-to-End Automatic Speech Recognition for Mandarin-English Code-Switching","date":"2021-12-19","arxiv_id":"2112.10202","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-turn-rnn-t-for-streaming-recognition-of","title":"Multi-turn RNN-T for streaming recognition of multi-party speech","date":"2021-12-19","arxiv_id":"2112.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-prompts-towards-memory-and-compute","title":"Prompt Tuning GPT-2 language model for parameter-efficient domain adaptation of ASR systems","date":"2021-12-16","arxiv_id":"2112.08718","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-hybrid-ctc-attention-end-to-end","title":"Improving Hybrid CTC/Attention End-to-end Speech Recognition with Pretrained Acoustic and Language Model","date":"2021-12-14","arxiv_id":"2112.07254","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-neural-voice-camouflage-1","title":"Real-Time Neural Voice Camouflage","date":"2021-12-14","arxiv_id":"2112.07076","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustifying-automatic-speech-recognition-by","title":"Robustifying automatic speech recognition by extracting slowly varying features","date":"2021-12-14","arxiv_id":"2112.07400","repositories_listed":0,"syntology":null},{"url":null,"slug":"pm-mmut-boosted-phone-mask-data-augmentation","title":"PM-MMUT: Boosted Phone-Mask Data Augmentation using Multi-Modeling Unit Training for Phonetic-Reduction-Robust E2E Speech Recognition","date":"2021-12-13","arxiv_id":"2112.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-code-switching-language-modeling","title":"Improving Code-switching Language Modeling with Artificially Generated Texts using Cycle-consistent Adversarial Networks","date":"2021-12-12","arxiv_id":"2112.06327","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-on-noisy-speech","title":"Improving Speech Recognition on Noisy Speech via Speech Enhancement with Multi-Discriminators CycleGAN","date":"2021-12-12","arxiv_id":"2112.06309","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-great-multi-lingual-teacher-with","title":"Building a great multi-lingual teacher with sparsely-gated mixture of experts for speech recognition","date":"2021-12-10","arxiv_id":"2112.05820","repositories_listed":0,"syntology":null},{"url":null,"slug":"directed-speech-separation-for-automatic","title":"Directed Speech Separation for Automatic Speech Recognition of Long Form Conversational Speech","date":"2021-12-10","arxiv_id":"2112.05863","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-boundary-between-asr-and-nlu","title":"Revisiting the Boundary between ASR and NLU in the Age of Conversational Dialog Systems","date":"2021-12-10","arxiv_id":"2112.05842","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-self-learning-with-multiple","title":"Sequence-level self-learning with multiple hypotheses","date":"2021-12-10","arxiv_id":"2112.05826","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-native-american-english-speech","title":"A study on native American English speech recognition by Indian listeners with varying word familiarity level","date":"2021-12-08","arxiv_id":"2112.04151","repositories_listed":0,"syntology":null},{"url":null,"slug":"bbs-kws-the-mandarin-keyword-spotting-system","title":"BBS-KWS:The Mandarin Keyword Spotting System Won the Video Keyword Wakeup Challenge","date":"2021-12-03","arxiv_id":"2112.01757","repositories_listed":0,"syntology":null},{"url":null,"slug":"blackbox-untargeted-adversarial-testing-of","title":"Catch Me If You Can: Blackbox Adversarial Attacks on Automatic Speech Recognition using Frequency Masking","date":"2021-12-03","arxiv_id":"2112.01821","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-higher-order-minkowski-loss-for-improved","title":"A higher order Minkowski loss for improved prediction ability of acoustic model in ASR","date":"2021-12-02","arxiv_id":"2112.01023","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mixture-of-expert-based-deep-neural-network","title":"A Mixture of Expert Based Deep Neural Network for Improved ASR","date":"2021-12-02","arxiv_id":"2112.01025","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-landscape-dependent-self-adjusting","title":"Loss Landscape Dependent Self-Adjusting Learning Rates in Decentralized Stochastic Gradient Descent","date":"2021-12-02","arxiv_id":"2112.01433","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experiment-on-speech-to-text-translation","title":"An Experiment on Speech-to-Text Translation Systems for Manipuri to English on Low Resource Setting","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-hybrid-architectures-for","title":"An Investigation of Hybrid architectures for Low Resource Multilingual Speech Recognition system in Indian context","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-manipuri-tones-in-manito-a-tonal","title":"Analysis of Manipuri Tones in ManiTo: A Tonal Contrast Database","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ie-cps-lexicon-an-automatic-speech","title":"IE-CPS Lexicon: An Automatic Speech Recognition Oriented Indian-English Pronunciation Dictionary","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-sinhala-speech-recognition-through","title":"Improve Sinhala Speech Recognition Through e2e LF-MMI Model","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-lexical-skills-from-oral-reading","title":"Predicting lexical skills from oral reading with acoustic measures","date":"2021-12-01","arxiv_id":"2112.00635","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-t-transducer-for-text-to-speech-and","title":"Speech-T: Transducer for Text to Speech and Beyond","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/do-we-still-need-automatic-speech-recognition","slug":"do-we-still-need-automatic-speech-recognition","title":"Do We Still Need Automatic Speech Recognition for Spoken Language Understanding?","date":"2021-11-29","arxiv_id":"2111.14842","repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-noise-suppression-losses-on-speech","title":"Effect of noise suppression losses on speech distortion and ASR performance","date":"2021-11-23","arxiv_id":"2111.11606","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-multi-speaker-asr-using-3d","title":"Multi-Channel Multi-Speaker ASR Using 3D Spatial Feature","date":"2021-11-22","arxiv_id":"2111.11023","repositories_listed":0,"syntology":null},{"url":null,"slug":"capitalization-and-punctuation-restoration-a","title":"Capitalization and Punctuation Restoration: a Survey","date":"2021-11-21","arxiv_id":"2111.10746","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-spoken-keyword-spotting-an-overview","title":"Deep Spoken Keyword Spotting: An Overview","date":"2021-11-20","arxiv_id":"2111.10592","repositories_listed":0,"syntology":null},{"url":null,"slug":"switching-independent-vector-analysis-and-its","title":"Switching Independent Vector Analysis and Its Extension to Blind and Spatially Guided Convolutional Beamforming Algorithms","date":"2021-11-20","arxiv_id":"2111.10574","repositories_listed":0,"syntology":null},{"url":null,"slug":"lattention-lattice-attention-in-asr-rescoring","title":"Lattention: Lattice-attention in ASR rescoring","date":"2021-11-19","arxiv_id":"2111.10157","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-conformer-based-asr-frontend-for-joint","title":"A Conformer-based ASR Frontend for Joint Acoustic Echo Cancellation, Speech Enhancement and Speech Separation","date":"2021-11-18","arxiv_id":"2111.09935","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-measuring-fairness-in-speech","title":"Towards Measuring Fairness in Speech Recognition: Casual Conversations Dataset Transcriptions","date":"2021-11-18","arxiv_id":"2111.09983","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-end-to-end-capt-system-for-l2","title":"A Novel End-to-End CAPT System for L2 Children Learners","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-language-model-optimization-in","title":"Heterogeneous Language Model Optimization in Automatic Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multimodal-speech-recognition-by","title":"Improving Multimodal Speech Recognition by Data Augmentation and Speech Representations","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-spoken-language-understanding-systems-for","title":"On Spoken Language Understanding Systems for Low Resourced Languages","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-down-sampling-for-acoustic","title":"Progressive Down-Sampling for Acoustic Encoding","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-sql-parsing-error-correction-with","title":"Speech-to-SQL Parsing: Error Correction with Multi-modal Representations","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"two-front-ends-one-model-fusing-heterogeneous","title":"Two Front-Ends, One Model : Fusing Heterogeneous Speech Features for Low Resource ASR with Multilingual Pre-Training","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"who-are-we-talking-about-handling-person","title":"Who Are We Talking About? Handling Person Names in Speech Translation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-end-to-end-speech-recognition-1","title":"Attention based end to end Speech Recognition for Voice Search in Hindi and English","date":"2021-11-15","arxiv_id":"2111.10208","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-of-listener-perception-of","title":"Prediction of Listener Perception of Argumentative Speech in a Crowdsourced Dataset Using (Psycho-)Linguistic and Fluency Features","date":"2021-11-13","arxiv_id":"2111.07130","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-normalized-importance-sampling-for","title":"Self-Normalized Importance Sampling for Neural Language Modeling","date":"2021-11-11","arxiv_id":"2111.06310","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-asr-improves-zero-and-few-shot","title":"Scaling ASR Improves Zero and Few Shot Learning","date":"2021-11-10","arxiv_id":"2111.05948","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-attacks-for-automatic-speech","title":"Privacy attacks for automatic speech recognition acoustic models in a federated learning framework","date":"2021-11-06","arxiv_id":"2111.03777","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformer-based-hybrid-asr-system-for","title":"Conformer-based Hybrid ASR System for Switchboard Dataset","date":"2021-11-05","arxiv_id":"2111.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-transformer-transducer-for","title":"Context-Aware Transformer Transducer for Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03250","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-speech-recognition-leveraging","title":"Effective Cross-Utterance Language Modeling for Conversational Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03333","repositories_listed":0,"syntology":null},{"url":"/paper/a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","slug":"a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","title":"A Fine-tuned Wav2vec 2.0/HuBERT Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding","date":"2021-11-04","arxiv_id":"2111.02735","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-for-air-traffic-control","title":"Speech recognition for air traffic control via feature learning and end-to-end training","date":"2021-11-04","arxiv_id":"2111.02654","repositories_listed":0,"syntology":null},{"url":null,"slug":"stc-speaker-recognition-systems-for-the-nist","title":"STC speaker recognition systems for the NIST SRE 2021","date":"2021-11-03","arxiv_id":"2111.02298","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advances-in-end-to-end-automatic","title":"Recent Advances in End-to-End Automatic Speech Recognition","date":"2021-11-02","arxiv_id":"2111.01690","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-data-relabeling-for-robust-and","title":"Collaborative Data Relabeling for Robust and Diverse Voice Apps Recommendation in Intelligent Personal Assistants","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-punctuation-restoration-for","title":"Comprehensive Punctuation Restoration for English and Polish","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"indic-languages-automatic-speech-recognition","title":"Indic Languages Automatic Speech Recognition using Meta-Learning Approach","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-transduction-with-graph-based","title":"Sequence Transduction with Graph-based Supervision","date":"2021-11-01","arxiv_id":"2111.01272","repositories_listed":0,"syntology":null},{"url":null,"slug":"snri-target-training-for-joint-speech","title":"SNRi Target Training for Joint Speech Enhancement and Recognition","date":"2021-11-01","arxiv_id":"2111.00764","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-technology-for-everyone-automatic-1","title":"Speech Technology for Everyone: Automatic Speech Recognition for Non-Native English","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-query-auto-completion","title":"Voice Query Auto Completion","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-attention-conformer-for-context","title":"Cross-attention conformer for context modeling in speech enhancement for ASR","date":"2021-10-30","arxiv_id":"2111.00127","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-conditioning-of-acoustic-models-using","title":"Speaker conditioning of acoustic models using affine transformation for multi-speaker speech recognition","date":"2021-10-30","arxiv_id":"2111.00320","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-asr-outputs-in-joint-training-for","title":"Fusing ASR Outputs in Joint Training for Speech Emotion Recognition","date":"2021-10-29","arxiv_id":"2110.15684","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-noise-robustness-of-contrastive","title":"Improving Noise Robustness of Contrastive Speech Representation Learning with Speech Reconstruction","date":"2021-10-28","arxiv_id":"2110.15430","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-decentralized-distributed","title":"Asynchronous Decentralized Distributed Training of Acoustic Models","date":"2021-10-21","arxiv_id":"2110.11199","repositories_listed":0,"syntology":null}],"record_sha256":"74645ef39de30fb130ae4a9ba89b215c9a7ad041b220854794eb73698deb4b17","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}