{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/20","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":32,"rows_per_page":100,"rows":[1901,2000],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/19","next":"/task/automatic-speech-recognition-2/papers/21","papers":[{"url":null,"slug":"end-to-end-multi-speaker-asr-with-independent","title":"End-to-End Multi-speaker ASR with Independent Vector Analysis","date":"2022-04-01","arxiv_id":"2204.00218","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-sequence-intermediate-conditioning-for","title":"Alternate Intermediate Conditioning with Syllable-level and Character-level Targets for Japanese ASR","date":"2022-04-01","arxiv_id":"2204.00175","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-rnn-t-with-semantic-decoder-for","title":"Multi-task RNN-T with Semantic Decoder for Streamable Spoken Language Understanding","date":"2022-04-01","arxiv_id":"2204.00558","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-speech-emotion-recognition","title":"Probing Speech Emotion Recognition Transformers for Linguistic Knowledge","date":"2022-04-01","arxiv_id":"2204.00400","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-data-augmentation-for-low","title":"Text-To-Speech Data Augmentation for Low Resource Speech Recognition","date":"2022-04-01","arxiv_id":"2204.00291","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-cross-lingual-aphasia-detection","title":"Zero-Shot Cross-lingual Aphasia Detection using Automatic Speech Recognition","date":"2022-04-01","arxiv_id":"2204.00448","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-speaker-attributed","title":"A Comparative Study on Speaker-attributed Automatic Speech Recognition in Multi-party Meetings","date":"2022-03-31","arxiv_id":"2203.16834","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-factors-affecting-usefulness-of","title":"Analyzing the factors affecting usefulness of Self-Supervised Pre-trained Representations for Speech Recognition","date":"2022-03-31","arxiv_id":"2203.16973","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectiveness-of-text-to-speech-pseudo-labels","title":"Effectiveness of text to speech pseudo labels for forced alignment and cross lingual pretrained models for low resource speech recognition","date":"2022-03-31","arxiv_id":"2203.16823","repositories_listed":0,"syntology":null},{"url":null,"slug":"importance-of-different-temporal-modulations","title":"Importance of Different Temporal Modulations of Speech: A Tale of Two Perspectives","date":"2022-03-31","arxiv_id":"2204.00065","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-efficient-training-of-rnn-transducer","title":"Memory-Efficient Training of RNN-Transducer with Sampled Softmax","date":"2022-03-31","arxiv_id":"2203.16868","repositories_listed":0,"syntology":null},{"url":"/paper/open-source-magicdata-ramc-a-rich-annotated","slug":"open-source-magicdata-ramc-a-rich-annotated","title":"Open Source MagicData-RAMC: A Rich Annotated Mandarin Conversational(RAMC) Speech Dataset","date":"2022-03-31","arxiv_id":"2203.16844","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switched-and-code-mixed-speech","title":"Code Switched and Code Mixed Speech Recognition for Indic languages","date":"2022-03-30","arxiv_id":"2203.16578","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-for-indic","title":"Improving Speech Recognition for Indic Languages using Language Model","date":"2022-03-30","arxiv_id":"2203.16595","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-word-error-rate-a-good-evaluation-metric","title":"Is Word Error Rate a good evaluation metric for Speech Recognition in Indic Languages?","date":"2022-03-30","arxiv_id":"2203.16601","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-latency-for-ctc-based-streaming","title":"Dynamic Latency for CTC-Based Streaming Automatic Speech Recognition With Emformer","date":"2022-03-29","arxiv_id":"2203.15613","repositories_listed":0,"syntology":null},{"url":null,"slug":"frequency-directional-attention-model-for","title":"Frequency-Directional Attention Model for Multilingual Automatic Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15473","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-of-deep-neural","title":"Improving Generalization of Deep Neural Network Acoustic Models with Length Perturbation and N-best Based Label Smoothing","date":"2022-03-29","arxiv_id":"2203.15176","repositories_listed":0,"syntology":null},{"url":null,"slug":"mel-frequency-spectral-domain-defenses","title":"Mel Frequency Spectral Domain Defenses against Adversarial Attacks on Speech Recognition Systems","date":"2022-03-29","arxiv_id":"2203.15283","repositories_listed":0,"syntology":null},{"url":null,"slug":"short-term-word-learning-in-a-dynamically","title":"Short-Term Word-Learning in a Dynamically Changing Environment","date":"2022-03-29","arxiv_id":"2203.15404","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-dataset-on-acoustic-models-for","title":"Impact of Dataset on Acoustic Models for Automatic Speech Recognition","date":"2022-03-25","arxiv_id":"2203.13590","repositories_listed":0,"syntology":null},{"url":null,"slug":"computing-optimal-location-of-microphone-for","title":"Computing Optimal Location of Microphone for Improved Speech Recognition","date":"2022-03-24","arxiv_id":"2203.13259","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangleing-content-and-fine-grained","title":"Disentangleing Content and Fine-grained Prosody Information via Hybrid ASR Bottleneck Features for Voice Conversion","date":"2022-03-24","arxiv_id":"2203.12813","repositories_listed":0,"syntology":null},{"url":null,"slug":"lahjoita-puhetta-a-large-scale-corpus-of","title":"Lahjoita puhetta -- a large-scale corpus of spoken Finnish with some benchmarks","date":"2022-03-24","arxiv_id":"2203.12906","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-text-to-speech-pipeline-evaluation","title":"A Text-to-Speech Pipeline, Evaluation Methodology, and Initial Fine-Tuning Results for Child Speech Synthesis","date":"2022-03-22","arxiv_id":"2203.11562","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-robust-spoken-language-understanding","title":"Building Robust Spoken Language Understanding by Cross Attention between Phoneme Sequence and ASR Hypothesis","date":"2022-03-22","arxiv_id":"2203.12067","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-label-is-better-than-human-label","title":"Pseudo Label Is Better Than Human Label","date":"2022-03-22","arxiv_id":"2203.12668","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-cross-domain-acoustic-to","title":"Exploiting Cross Domain Acoustic-to-articulatory Inverted Features For Disordered Speech Recognition","date":"2022-03-19","arxiv_id":"2203.10274","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-representative-subset-selection-for","title":"Representative Subset Selection for Efficient Fine-Tuning in Self-Supervised Speech Recognition","date":"2022-03-18","arxiv_id":"2203.09829","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-of-speech-intelligibility-with-dnn","title":"Prediction of speech intelligibility with DNN-based performance measures","date":"2022-03-17","arxiv_id":"2203.09148","repositories_listed":0,"syntology":null},{"url":null,"slug":"whither-the-priors-for-vocal-interactivity","title":"Whither the Priors for (Vocal) Interactivity?","date":"2022-03-16","arxiv_id":"2203.08578","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-modification-based-data-augmentation","title":"Spectral Modification Based Data Augmentation For Improving End-to-End ASR For Children's Speech","date":"2022-03-13","arxiv_id":"2203.06600","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-streaming-asr-with","title":"Transformer-based Streaming ASR with Cumulative Attention","date":"2022-03-11","arxiv_id":"2203.05736","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacks-as-defenses-designing-robust-audio","title":"Attacks as Defenses: Designing Robust Audio CAPTCHAs Using Attacks on Automatic Speech Recognition Systems","date":"2022-03-10","arxiv_id":"2203.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-practical-framework-for-multi-domain-speech","title":"A practical framework for multi-domain speech recognition and an instance sampling method to neural language modeling","date":"2022-03-09","arxiv_id":"2203.04767","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-french-speech-recognition-system-for","title":"Which French speech recognition system for assistant robots?","date":"2022-03-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-conformer-based-acoustic-model-for-robust","title":"A Conformer Based Acoustic Model for Robust Automatic Speech Recognition","date":"2022-03-01","arxiv_id":"2203.00725","repositories_listed":0,"syntology":null},{"url":null,"slug":"extended-graph-temporal-classification-for","title":"Extended Graph Temporal Classification for Multi-Speaker End-to-End ASR","date":"2022-03-01","arxiv_id":"2203.00232","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-the-impact-of-individual-domain","title":"Measuring the Impact of Individual Domain Factors in Self-Supervised Pre-Training","date":"2022-03-01","arxiv_id":"2203.00648","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-text-inputs-for-training-and","title":"Integrating Text Inputs For Training and Adapting RNN Transducer ASR Models","date":"2022-02-26","arxiv_id":"2202.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-multilingual-models-for-automatic","title":"A Survey of Multilingual Models for Automatic Speech Recognition","date":"2022-02-25","arxiv_id":"2202.12576","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-technology-practitioners-as-language","title":"Language technology practitioners as language managers: arbitrating data bias and predictive bias in ASR","date":"2022-02-25","arxiv_id":"2202.12603","repositories_listed":0,"syntology":null},{"url":null,"slug":"ask2mask-guided-data-selection-for-masked-1","title":"Ask2Mask: Guided Data Selection for Masked Speech Modeling","date":"2022-02-24","arxiv_id":"2202.12719","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-better-meta-initialization-with-task","title":"Towards Better Meta-Initialization with Task Augmentation for Kindergarten-aged Speech Recognition","date":"2022-02-24","arxiv_id":"2202.12326","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-speaker-anonymization","title":"Differentially Private Speaker Anonymization","date":"2022-02-23","arxiv_id":"2202.11823","repositories_listed":0,"syntology":null},{"url":null,"slug":"korean-tokenization-for-beam-search-rescoring","title":"Korean Tokenization for Beam Search Rescoring in Speech Recognition","date":"2022-02-22","arxiv_id":"2203.03583","repositories_listed":0,"syntology":null},{"url":null,"slug":"vadoi-voice-activity-detection-overlapping","title":"VADOI:Voice-Activity-Detection Overlapping Inference For End-to-end Long-form Speech Recognition","date":"2022-02-22","arxiv_id":"2202.10593","repositories_listed":0,"syntology":null},{"url":null,"slug":"r-g2p-evaluating-and-enhancing-robustness-of","title":"r-G2P: Evaluating and Enhancing Robustness of Grapheme to Phoneme Conversion by Controlled noise introducing and Contextual information incorporation","date":"2022-02-21","arxiv_id":"2202.11194","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-adaptation-using-spectro-temporal","title":"Speaker Adaptation Using Spectro-Temporal Deep Features for Dysarthric and Elderly Speech Recognition","date":"2022-02-21","arxiv_id":"2202.10290","repositories_listed":0,"syntology":null},{"url":"/paper/punctuation-restoration","slug":"punctuation-restoration","title":"SemEval 2022 Task 12: Symlink- Linking Mathematical Symbols to their Descriptions","date":"2022-02-19","arxiv_id":"2202.09695","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-of-low-resource-target","title":"Domain Adaptation of low-resource Target-Domain models using well-trained ASR Conformer Models","date":"2022-02-18","arxiv_id":"2202.09167","repositories_listed":0,"syntology":null},{"url":null,"slug":"beach-to-bitch-inadvertent-unsafe","title":"'Beach' to 'Bitch': Inadvertent Unsafe Transcription of Kids' Content on YouTube","date":"2022-02-17","arxiv_id":"2203.04837","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-closed-model-adversarial-examples","title":"Mitigating Closed-model Adversarial Examples with Bayesian Neural Modeling for Enhanced End-to-End Speech Recognition","date":"2022-02-17","arxiv_id":"2202.08532","repositories_listed":0,"syntology":null},{"url":null,"slug":"mlp-asr-sequence-length-agnostic-all-mlp","title":"MLP-ASR: Sequence-length agnostic all-MLP architectures for speech recognition","date":"2022-02-17","arxiv_id":"2202.08456","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-speech-recognition-by-learning","title":"Conversational Speech Recognition By Learning Conversation-level Characteristics","date":"2022-02-16","arxiv_id":"2202.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-transfer-from-large-scale","title":"Knowledge Transfer from Large-scale Pretrained Language Models to End-to-end Speech Recognizers","date":"2022-02-16","arxiv_id":"2202.07894","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-style-training-for-south-african-call","title":"Multi-style Training for South African Call Centre Audio","date":"2022-02-15","arxiv_id":"2202.07219","repositories_listed":0,"syntology":null},{"url":null,"slug":"saving-rnn-computations-with-a-neuron-level","title":"Saving RNN Computations with a Neuron-Level Fuzzy Memoization Scheme","date":"2022-02-14","arxiv_id":"2202.06563","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-depression-classification-using","title":"Multimodal Depression Classification Using Articulatory Coordination Features And Hierarchical Attention Based Text Embeddings","date":"2022-02-13","arxiv_id":"2202.06238","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-step-approach-to-leverage-contextual","title":"A two-step approach to leverage contextual data: speech recognition in air-traffic communications","date":"2022-02-08","arxiv_id":"2202.03725","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-asr-for-stuttered-speech-with","title":"Enhancing ASR for Stuttered Speech with Limited Data Using Detect and Pass","date":"2022-02-08","arxiv_id":"2202.05396","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyphonic-pitch-detection-with-convolutional","title":"Polyphonic pitch detection with convolutional recurrent neural networks","date":"2022-02-04","arxiv_id":"2202.02115","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-cuhk-tencent-speaker-diarization-system","title":"The CUHK-TENCENT speaker diarization system for the ICASSP 2022 multi-channel multi-party meeting transcription challenge","date":"2022-02-04","arxiv_id":"2202.01986","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-speech-recognition-and-audio-captioning","title":"Joint Speech Recognition and Audio Captioning","date":"2022-02-03","arxiv_id":"2202.01405","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-royalflush-system-of-speech-recognition","title":"The RoyalFlush System of Speech Recognition for M2MeT Challenge","date":"2022-02-03","arxiv_id":"2202.01614","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-aware-end-to-end-neural-diarization","title":"ASR-Aware End-to-end Neural Diarization","date":"2022-02-02","arxiv_id":"2202.01286","repositories_listed":0,"syntology":null},{"url":null,"slug":"error-correction-in-asr-using-sequence-to","title":"Error Correction in ASR using Sequence-to-Sequence Models","date":"2022-02-02","arxiv_id":"2202.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"rescorebert-discriminative-speech-recognition","title":"RescoreBERT: Discriminative Speech Recognition Rescoring with BERT","date":"2022-02-02","arxiv_id":"2202.01094","repositories_listed":0,"syntology":null},{"url":null,"slug":"bea-base-a-benchmark-for-asr-of-spontaneous","title":"BEA-Base: A Benchmark for ASR of Spontaneous Hungarian","date":"2022-02-01","arxiv_id":"2202.00601","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-dependencies-in-adversarial-attacks","title":"Language Dependencies in Adversarial Attacks on Speech Recognition Systems","date":"2022-02-01","arxiv_id":"2202.00399","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-automatic-speech-recognition","title":"Visualizing Automatic Speech Recognition -- Means for a Better Understanding?","date":"2022-02-01","arxiv_id":"2202.00673","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-language-context-confusion-for-end","title":"Reducing language context confusion for end-to-end code-switching automatic speech recognition","date":"2022-01-28","arxiv_id":"2201.12155","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentiment-aware-automatic-speech-recognition","title":"Sentiment-Aware Automatic Speech Recognition pre-training for enhanced Speech Emotion Recognition","date":"2022-01-27","arxiv_id":"2201.11826","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-dysarthric-speech-using-multi","title":"Synthesizing Dysarthric Speech Using Multi-talker TTS for Dysarthric Speech Recognition","date":"2022-01-27","arxiv_id":"2201.11571","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-decoder-transformer-for-end-to-end","title":"On the Effectiveness of Pinyin-Character Dual-Decoding for End-to-End Mandarin Chinese ASR","date":"2022-01-26","arxiv_id":"2201.10792","repositories_listed":0,"syntology":null},{"url":"/paper/the-norwegian-parliamentary-speech-corpus","slug":"the-norwegian-parliamentary-speech-corpus","title":"The Norwegian Parliamentary Speech Corpus","date":"2022-01-26","arxiv_id":"2201.10881","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-non-autoregressive-end-to-end","title":"Improving non-autoregressive end-to-end speech recognition with pre-trained acoustic and language models","date":"2022-01-25","arxiv_id":"2201.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-fusion-of-acoustic-and-text","title":"Improving the fusion of acoustic and text representations in RNN-T","date":"2022-01-25","arxiv_id":"2201.10240","repositories_listed":0,"syntology":null},{"url":null,"slug":"run-and-back-stitch-search-novel-block","title":"Run-and-back stitch search: novel block synchronous decoding for streaming encoder-decoder ASR","date":"2022-01-25","arxiv_id":"2201.10190","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-video-front-ends-for-audio","title":"Transformer-Based Video Front-Ends for Audio-Visual Speech Recognition for Single and Multi-Person Video","date":"2022-01-25","arxiv_id":"2201.10439","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-noise-robust-self-supervised-pre-training","title":"A Noise-Robust Self-supervised Pre-training Model Based Speech Representation Learning for Automatic Speech Recognition","date":"2022-01-22","arxiv_id":"2201.08930","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-bad-are-artifacts-analyzing-the-impact-of","title":"How Bad Are Artifacts?: Analyzing the Impact of Speech Enhancement Errors on ASR","date":"2022-01-18","arxiv_id":"2201.06685","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-and-automatic-speech-recognition","title":"Human and Automatic Speech Recognition Performance on German Oral History Interviews","date":"2022-01-18","arxiv_id":"2201.06841","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-textless-spoken-question-answering-with","title":"DUAL: Textless Spoken Question Answering with Speech Discrete Unit Adaptive Learning","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"red-ace-robust-error-detection-for-asr-using","title":"RED-ACE: Robust Error Detection for ASR using Confidence Embeddings","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-progress-in-the-cuhk-dysarthric-speech","title":"Recent Progress in the CUHK Dysarthric Speech Recognition System","date":"2022-01-15","arxiv_id":"2201.05845","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-enhance-or-not-neural-network","title":"Learning to Enhance or Not: Neural Network-Based Switching of Enhanced and Observed Signals for Overlapping Speech Recognition","date":"2022-01-11","arxiv_id":"2201.03881","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-likelihood-ratio-based-domain-adaptation","title":"A Likelihood Ratio based Domain Adaptation Method for E2E Models","date":"2022-01-10","arxiv_id":"2201.03655","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-asr-post-processing-system-for","title":"Cross-Modal ASR Post-Processing System for Error Correction and Utterance Rejection","date":"2022-01-10","arxiv_id":"2201.03313","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-sql-towards-speech-driven-sql-query","title":"Speech-to-SQL: Towards Speech-driven SQL Query Generation From Natural Language Question","date":"2022-01-04","arxiv_id":"2201.01209","repositories_listed":0,"syntology":null},{"url":null,"slug":"tencent-mvse-a-large-scale-benchmark-dataset","title":"Tencent-MVSE: A Large-Scale Benchmark Dataset for Multi-Modal Video Similarity Evaluation","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-dialect-arabic-speech-recognition","title":"Multi-Dialect Arabic Speech Recognition","date":"2021-12-25","arxiv_id":"2112.14678","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-augmentation-based-consistency","title":"Multi-Variant Consistency based Self-supervised Learning for Robust Automatic Speech Recognition","date":"2021-12-23","arxiv_id":"2112.12522","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-quality-and-pitch-features-in","title":"Voice Quality and Pitch Features in Transformer-Based Speech Recognition","date":"2021-12-21","arxiv_id":"2112.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-knowledge-in-end-to-end-automatic","title":"Integrating Knowledge in End-to-End Automatic Speech Recognition for Mandarin-English Code-Switching","date":"2021-12-19","arxiv_id":"2112.10202","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-turn-rnn-t-for-streaming-recognition-of","title":"Multi-turn RNN-T for streaming recognition of multi-party speech","date":"2021-12-19","arxiv_id":"2112.10200","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-prompts-towards-memory-and-compute","title":"Prompt Tuning GPT-2 language model for parameter-efficient domain adaptation of ASR systems","date":"2021-12-16","arxiv_id":"2112.08718","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-hybrid-ctc-attention-end-to-end","title":"Improving Hybrid CTC/Attention End-to-end Speech Recognition with Pretrained Acoustic and Language Model","date":"2021-12-14","arxiv_id":"2112.07254","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-neural-voice-camouflage-1","title":"Real-Time Neural Voice Camouflage","date":"2021-12-14","arxiv_id":"2112.07076","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustifying-automatic-speech-recognition-by","title":"Robustifying automatic speech recognition by extracting slowly varying features","date":"2021-12-14","arxiv_id":"2112.07400","repositories_listed":0,"syntology":null}],"record_sha256":"5840cffaf8db295b68c17cedac5d49d6a43b0d2fbf9d9ab7a7eed7fbfce9bed5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}