{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/34","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":34,"pages_in_order":58,"rows_per_page":100,"rows":[3301,3400],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/33","next":"/task/speech-recognition-1/papers/35","papers":[{"url":null,"slug":"a-likelihood-ratio-based-domain-adaptation","title":"A Likelihood Ratio based Domain Adaptation Method for E2E Models","date":"2022-01-10","arxiv_id":"2201.03655","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-asr-post-processing-system-for","title":"Cross-Modal ASR Post-Processing System for Error Correction and Utterance Rejection","date":"2022-01-10","arxiv_id":"2201.03313","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-pass-end-to-end-asr-model-compression","title":"Two-Pass End-to-End ASR Model Compression","date":"2022-01-08","arxiv_id":"2201.02741","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switching-text-augmentation-for","title":"Textual Data Augmentation for Arabic-English Code-Switching Speech Recognition","date":"2022-01-07","arxiv_id":"2201.02550","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-sql-towards-speech-driven-sql-query","title":"Speech-to-SQL: Towards Speech-driven SQL Query Generation From Natural Language Question","date":"2022-01-04","arxiv_id":"2201.01209","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-natural-language-processing-recent","title":"Robust Natural Language Processing: Recent Advances, Challenges, and Future Directions","date":"2022-01-03","arxiv_id":"2201.00768","repositories_listed":0,"syntology":null},{"url":null,"slug":"tencent-mvse-a-large-scale-benchmark-dataset","title":"Tencent-MVSE: A Large-Scale Benchmark Dataset for Multi-Modal Video Similarity Evaluation","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-attention-augmented-transformer","title":"Temporal Attention Augmented Transformer Hawkes Process","date":"2021-12-29","arxiv_id":"2112.14472","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-using-deep-acoustic","title":"Bridging the Gap: Using Deep Acoustic Representations to Learn Grounded Language from Percepts and Raw Speech","date":"2021-12-27","arxiv_id":"2112.13758","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-dialect-arabic-speech-recognition","title":"Multi-Dialect Arabic Speech Recognition","date":"2021-12-25","arxiv_id":"2112.14678","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-based-consistency","title":"Multi-Variant Consistency based Self-supervised Learning for Robust Automatic Speech Recognition","date":"2021-12-23","arxiv_id":"2112.12522","repositories_listed":0,"syntology":null},{"url":null,"slug":"tod-da-towards-boosting-the-robustness-of","title":"TOD-DA: Towards Boosting the Robustness of Task-oriented Dialogue Modeling on Spoken Conversations","date":"2021-12-23","arxiv_id":"2112.12441","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicemoji-a-novel-on-device-pipeline-for","title":"VoiceMoji: A Novel On-Device Pipeline for Seamless Emoji Insertion in Dictation","date":"2021-12-22","arxiv_id":"2112.12028","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-quality-and-pitch-features-in","title":"Voice Quality and Pitch Features in Transformer-Based Speech Recognition","date":"2021-12-21","arxiv_id":"2112.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"load-balanced-gather-scatter-patterns-for","title":"Load-balanced Gather-scatter Patterns for Sparse Deep Neural Networks","date":"2021-12-20","arxiv_id":"2112.10898","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-knowledge-in-end-to-end-automatic","title":"Integrating Knowledge in End-to-End Automatic Speech Recognition for Mandarin-English Code-Switching","date":"2021-12-19","arxiv_id":"2112.10202","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-densely-connected","title":"Investigation of Densely Connected Convolutional Networks with Domain Adversarial Learning for Noise Robust Speech Recognition","date":"2021-12-19","arxiv_id":"2112.10108","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-turn-rnn-t-for-streaming-recognition-of","title":"Multi-turn RNN-T for streaming recognition of multi-party speech","date":"2021-12-19","arxiv_id":"2112.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-singular-riemannian-geometry-approach-to-1","title":"A singular Riemannian geometry approach to Deep Neural Networks I. Theoretical foundations","date":"2021-12-17","arxiv_id":"2201.09656","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-prompts-towards-memory-and-compute","title":"Prompt Tuning GPT-2 language model for parameter-efficient domain adaptation of ASR systems","date":"2021-12-16","arxiv_id":"2112.08718","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-hybrid-ctc-attention-end-to-end","title":"Improving Hybrid CTC/Attention End-to-end Speech Recognition with Pretrained Acoustic and Language Model","date":"2021-12-14","arxiv_id":"2112.07254","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-neural-voice-camouflage-1","title":"Real-Time Neural Voice Camouflage","date":"2021-12-14","arxiv_id":"2112.07076","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustifying-automatic-speech-recognition-by","title":"Robustifying automatic speech recognition by extracting slowly varying features","date":"2021-12-14","arxiv_id":"2112.07400","repositories_listed":0,"syntology":null},{"url":null,"slug":"pm-mmut-boosted-phone-mask-data-augmentation","title":"PM-MMUT: Boosted Phone-Mask Data Augmentation using Multi-Modeling Unit Training for Phonetic-Reduction-Robust E2E Speech Recognition","date":"2021-12-13","arxiv_id":"2112.06721","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-code-switching-language-modeling","title":"Improving Code-switching Language Modeling with Artificially Generated Texts using Cycle-consistent Adversarial Networks","date":"2021-12-12","arxiv_id":"2112.06327","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-on-noisy-speech","title":"Improving Speech Recognition on Noisy Speech via Speech Enhancement with Multi-Discriminators CycleGAN","date":"2021-12-12","arxiv_id":"2112.06309","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-great-multi-lingual-teacher-with","title":"Building a great multi-lingual teacher with sparsely-gated mixture of experts for speech recognition","date":"2021-12-10","arxiv_id":"2112.05820","repositories_listed":0,"syntology":null},{"url":null,"slug":"directed-speech-separation-for-automatic","title":"Directed Speech Separation for Automatic Speech Recognition of Long Form Conversational Speech","date":"2021-12-10","arxiv_id":"2112.05863","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-boundary-between-asr-and-nlu","title":"Revisiting the Boundary between ASR and NLU in the Age of Conversational Dialog Systems","date":"2021-12-10","arxiv_id":"2112.05842","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-self-learning-with-multiple","title":"Sequence-level self-learning with multiple hypotheses","date":"2021-12-10","arxiv_id":"2112.05826","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-e2e-asr-models-ready-for-an-industrial","title":"Are E2E ASR models ready for an industrial usage?","date":"2021-12-09","arxiv_id":"2112.12572","repositories_listed":0,"syntology":null},{"url":null,"slug":"lipsound2-self-supervised-pre-training-for","title":"LipSound2: Self-Supervised Pre-Training for Lip-to-Speech Reconstruction and Lip Reading","date":"2021-12-09","arxiv_id":"2112.04748","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-native-american-english-speech","title":"A study on native American English speech recognition by Indian listeners with varying word familiarity level","date":"2021-12-08","arxiv_id":"2112.04151","repositories_listed":0,"syntology":null},{"url":null,"slug":"bbs-kws-the-mandarin-keyword-spotting-system","title":"BBS-KWS:The Mandarin Keyword Spotting System Won the Video Keyword Wakeup Challenge","date":"2021-12-03","arxiv_id":"2112.01757","repositories_listed":0,"syntology":null},{"url":null,"slug":"blackbox-untargeted-adversarial-testing-of","title":"Catch Me If You Can: Blackbox Adversarial Attacks on Automatic Speech Recognition using Frequency Masking","date":"2021-12-03","arxiv_id":"2112.01821","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-higher-order-minkowski-loss-for-improved","title":"A higher order Minkowski loss for improved prediction ability of acoustic model in ASR","date":"2021-12-02","arxiv_id":"2112.01023","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mixture-of-expert-based-deep-neural-network","title":"A Mixture of Expert Based Deep Neural Network for Improved ASR","date":"2021-12-02","arxiv_id":"2112.01025","repositories_listed":0,"syntology":null},{"url":null,"slug":"loss-landscape-dependent-self-adjusting","title":"Loss Landscape Dependent Self-Adjusting Learning Rates in Decentralized Stochastic Gradient Descent","date":"2021-12-02","arxiv_id":"2112.01433","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-speech-recognition-for-the","title":"An End-to-End Speech Recognition for the Nepali Language","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experiment-on-speech-to-text-translation","title":"An Experiment on Speech-to-Text Translation Systems for Manipuri to English on Low Resource Setting","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-hybrid-architectures-for","title":"An Investigation of Hybrid architectures for Low Resource Multilingual Speech Recognition system in Indian context","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-manipuri-tones-in-manito-a-tonal","title":"Analysis of Manipuri Tones in ManiTo: A Tonal Contrast Database","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ie-cps-lexicon-an-automatic-speech","title":"IE-CPS Lexicon: An Automatic Speech Recognition Oriented Indian-English Pronunciation Dictionary","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-microphone-position-measurement","title":"Impact of Microphone position Measurement Error on Multi Channel Distant Speech Recognition & Intelligibility","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-sinhala-speech-recognition-through","title":"Improve Sinhala Speech Recognition Through e2e LF-MMI Model","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"phone-based-keyword-spotting-for-transcribing","title":"Phone Based Keyword Spotting for Transcribing Very Low Resource Languages","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-lexical-skills-from-oral-reading","title":"Predicting lexical skills from oral reading with acoustic measures","date":"2021-12-01","arxiv_id":"2112.00635","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-t-transducer-for-text-to-speech-and","title":"Speech-T: Transducer for Text to Speech and Beyond","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/do-we-still-need-automatic-speech-recognition","slug":"do-we-still-need-automatic-speech-recognition","title":"Do We Still Need Automatic Speech Recognition for Spoken Language Understanding?","date":"2021-11-29","arxiv_id":"2111.14842","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-modeling-of-code-switched-and","title":"Joint Modeling of Code-Switched and Monolingual ASR via Conditional Factorization","date":"2021-11-29","arxiv_id":"2111.15016","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-precision-low-bit-quantization-of","title":"Mixed Precision Low-bit Quantization of Neural Network Language Models for Speech Recognition","date":"2021-11-29","arxiv_id":"2112.11438","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-precision-of-quantization-of","title":"Mixed Precision of Quantization of Transformer Language Models for Speech Recognition","date":"2021-11-29","arxiv_id":"2112.11540","repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-noise-suppression-losses-on-speech","title":"Effect of noise suppression losses on speech distortion and ASR performance","date":"2021-11-23","arxiv_id":"2111.11606","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-text-to-speech-with-untranscribed-1","title":"Guided-TTS: A Diffusion Model for Text-to-Speech via Classifier Guidance","date":"2021-11-23","arxiv_id":"2111.11755","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechmoe2-mixture-of-experts-model-with","title":"SpeechMoE2: Mixture-of-Experts Model with Improved Routing","date":"2021-11-23","arxiv_id":"2111.11831","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-multi-speaker-asr-using-3d","title":"Multi-Channel Multi-Speaker ASR Using 3D Spatial Feature","date":"2021-11-22","arxiv_id":"2111.11023","repositories_listed":0,"syntology":null},{"url":null,"slug":"capitalization-and-punctuation-restoration-a","title":"Capitalization and Punctuation Restoration: a Survey","date":"2021-11-21","arxiv_id":"2111.10746","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-spoken-keyword-spotting-an-overview","title":"Deep Spoken Keyword Spotting: An Overview","date":"2021-11-20","arxiv_id":"2111.10592","repositories_listed":0,"syntology":null},{"url":null,"slug":"switching-independent-vector-analysis-and-its","title":"Switching Independent Vector Analysis and Its Extension to Blind and Spatially Guided Convolutional Beamforming Algorithms","date":"2021-11-20","arxiv_id":"2111.10574","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-streaming-models-and-data","title":"A comparison of streaming models and data augmentation methods for robust speech recognition","date":"2021-11-19","arxiv_id":"2111.10043","repositories_listed":0,"syntology":null},{"url":null,"slug":"lattention-lattice-attention-in-asr-rescoring","title":"Lattention: Lattice-attention in ASR rescoring","date":"2021-11-19","arxiv_id":"2111.10157","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-transfer-learning-for","title":"Semi-supervised transfer learning for language expansion of end-to-end speech recognition models to low-resource languages","date":"2021-11-19","arxiv_id":"2111.10047","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-conformer-based-asr-frontend-for-joint","title":"A Conformer-based ASR Frontend for Joint Acoustic Echo Cancellation, Speech Enhancement and Speech Separation","date":"2021-11-18","arxiv_id":"2111.09935","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-measuring-fairness-in-speech","title":"Towards Measuring Fairness in Speech Recognition: Casual Conversations Dataset Transcriptions","date":"2021-11-18","arxiv_id":"2111.09983","repositories_listed":0,"syntology":null},{"url":null,"slug":"subject-enveloped-deep-sample-fuzzy-ensemble","title":"Subject Enveloped Deep Sample Fuzzy Ensemble Learning Algorithm of Parkinson's Speech Data","date":"2021-11-17","arxiv_id":"2111.09014","repositories_listed":0,"syntology":null},{"url":"/paper/the-people-s-speech-a-large-scale-diverse","slug":"the-people-s-speech-a-large-scale-diverse","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","date":"2021-11-17","arxiv_id":"2111.09344","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-end-to-end-capt-system-for-l2","title":"A Novel End-to-End CAPT System for L2 Children Learners","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-language-model-optimization-in","title":"Heterogeneous Language Model Optimization in Automatic Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-multimodal-speech-recognition-by","title":"Improving Multimodal Speech Recognition by Data Augmentation and Speech Representations","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-uni-modal-self-supervised-learning","title":"Leveraging Uni-Modal Self-Supervised Learning for Multimodal Audio-visual Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-speech-recognition-and-synthesis","title":"Modeling speech recognition and synthesis simultaneously: Encoding and decoding lexical and sublexical semantic information into speech with no access to speech data","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-spoken-language-understanding-systems-for","title":"On Spoken Language Understanding Systems for Low Resourced Languages","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-down-sampling-for-acoustic","title":"Progressive Down-Sampling for Acoustic Encoding","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-for-low-resource-languages","title":"Speech Synthesis for Low Resource Languages using Transliteration Enabled Transfer Learning","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-sql-parsing-error-correction-with","title":"Speech-to-SQL Parsing: Error Correction with Multi-modal Representations","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"two-front-ends-one-model-fusing-heterogeneous","title":"Two Front-Ends, One Model : Fusing Heterogeneous Speech Features for Low Resource ASR with Multilingual Pre-Training","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-speech-text-pre-training-for-speech","title":"Unified Speech-Text Pre-training for Speech Translation and Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-speech-enhancement-with-speech","title":"Unsupervised Speech Enhancement with speech recognition embedding and disentanglement losses","date":"2021-11-16","arxiv_id":"2111.08678","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-are-we-talking-about-handling-person","title":"Who Are We Talking About? Handling Person Names in Speech Translation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-end-to-end-speech-recognition-1","title":"Attention based end to end Speech Recognition for Voice Search in Hindi and English","date":"2021-11-15","arxiv_id":"2111.10208","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-for-speech-recognition-in","title":"Analysis of Data Augmentation Methods for Low-Resource Maltese ASR","date":"2021-11-15","arxiv_id":"2111.07793","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-unsupervised-and-supervised-training","title":"Joint Unsupervised and Supervised Training for Multilingual ASR","date":"2021-11-15","arxiv_id":"2111.08137","repositories_listed":0,"syntology":null},{"url":null,"slug":"binary-classification-of-spoken-words-with","title":"Binary classification of spoken words with passive phononic metamaterials","date":"2021-11-14","arxiv_id":"2111.08503","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-of-listener-perception-of","title":"Prediction of Listener Perception of Argumentative Speech in a Crowdsourced Dataset Using (Psycho-)Linguistic and Fluency Features","date":"2021-11-13","arxiv_id":"2111.07130","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-convolutional-neural-network-based-approach-1","title":"A Convolutional Neural Network Based Approach to Recognize Bangla Spoken Digits from Speech Signal","date":"2021-11-12","arxiv_id":"2111.06625","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-neural-networks-predict-dynamics-they","title":"Can neural networks predict dynamics they have never seen?","date":"2021-11-12","arxiv_id":"2111.06783","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-normalized-importance-sampling-for","title":"Self-Normalized Importance Sampling for Neural Language Modeling","date":"2021-11-11","arxiv_id":"2111.06310","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-asr-improves-zero-and-few-shot","title":"Scaling ASR Improves Zero and Few Shot Learning","date":"2021-11-10","arxiv_id":"2111.05948","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-aec-and-beamforming-with-double-talk","title":"Joint Neural AEC and Beamforming with Double-Talk Detection","date":"2021-11-09","arxiv_id":"2111.04904","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieving-speaker-information-from","title":"Retrieving Speaker Information from Personalized Acoustic Models for Speech Recognition","date":"2021-11-07","arxiv_id":"2111.04194","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-attacks-for-automatic-speech","title":"Privacy attacks for automatic speech recognition acoustic models in a federated learning framework","date":"2021-11-06","arxiv_id":"2111.03777","repositories_listed":0,"syntology":null},{"url":null,"slug":"conformer-based-hybrid-asr-system-for","title":"Conformer-based Hybrid ASR System for Switchboard Dataset","date":"2021-11-05","arxiv_id":"2111.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-transformer-transducer-for","title":"Context-Aware Transformer Transducer for Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03250","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-speech-recognition-leveraging","title":"Effective Cross-Utterance Language Modeling for Conversational Speech Recognition","date":"2021-11-05","arxiv_id":"2111.03333","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-of-frequency-time-attention","title":"Learning of Time-Frequency Attention Mechanism for Automatic Modulation Recognition","date":"2021-11-05","arxiv_id":"2111.03258","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-teacher-towards-better-knowledge","title":"Oracle Teacher: Leveraging Target Information for Better Knowledge Distillation of CTC Models","date":"2021-11-05","arxiv_id":"2111.03664","repositories_listed":0,"syntology":null},{"url":"/paper/a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","slug":"a-fine-tuned-wav2vec-2-0-hubert-benchmark-for","title":"A Fine-tuned Wav2vec 2.0/HuBERT Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding","date":"2021-11-04","arxiv_id":"2111.02735","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-for-air-traffic-control","title":"Speech recognition for air traffic control via feature learning and end-to-end training","date":"2021-11-04","arxiv_id":"2111.02654","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-can-improve-asr-in-very-low","title":"Voice Conversion Can Improve ASR in Very Low-Resource Settings","date":"2021-11-04","arxiv_id":"2111.02674","repositories_listed":0,"syntology":null},{"url":null,"slug":"stc-speaker-recognition-systems-for-the-nist","title":"STC speaker recognition systems for the NIST SRE 2021","date":"2021-11-03","arxiv_id":"2111.02298","repositories_listed":0,"syntology":null}],"record_sha256":"b4d698481cd3643ab900acdea586a6d5e3562b56a9b16cb94f76ec5d04b2226b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}