{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/37","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":37,"pages_in_order":65,"rows_per_page":100,"rows":[3601,3700],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/36","next":"/task/speech-recognition/papers/38","papers":[{"url":null,"slug":"employing-low-pass-filtered-temporal-speech","title":"Employing low-pass filtered temporal speech features for the training of ideal ratio mask in speech enhancement","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-low-resource-code-switching-data","title":"Exploiting Low-Resource Code-Switching Data to Mandarin-English Speech Recognition Systems","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-integration-of-e2e-asr-and","title":"Exploring the Integration of E2E ASR and Pronunciation Modeling for English Mispronunciation Detection","date":"2021-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-layer-wise-self-supervised","title":"Incremental Layer-wise Self-Supervised Learning for Efficient Speech Domain Adaptation On Device","date":"2021-10-01","arxiv_id":"2110.00155","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-technology-for-everyone-automatic","title":"Speech Technology for Everyone: Automatic Speech Recognition for Non-Native English with Transfer Learning","date":"2021-10-01","arxiv_id":"2110.00678","repositories_listed":0,"syntology":null},{"url":null,"slug":"spliceout-a-simple-and-efficient-audio","title":"SpliceOut: A Simple and Efficient Audio Augmentation Method","date":"2021-09-30","arxiv_id":"2110.00046","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-lottery-speech-recognition-made-ultra","title":"Audio Lottery: Speech Recognition Made Ultra-Lightweight, Noise-Robust, and Transferable","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-self-supervised-speech-pre","title":"Comparison of Self-Supervised Speech Pre-Training Methods on Flemish Dutch","date":"2021-09-29","arxiv_id":"2109.14357","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-sequence-to-sequence-networks","title":"Conditioning Sequence-to-sequence Networks with Learned Activations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-limited-adversarial","title":"Demystifying Limited Adversarial Transferability in Automatic Speech Recognition Systems","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ep-gan-unsupervised-federated-learning-with","title":"EP-GAN: Unsupervised Federated Learning with Expectation-Propagation Prior GAN","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"google-neural-network-models-for-edge-devices","title":"Google Neural Network Models for Edge Devices: Analyzing and Mitigating Machine Learning Inference Bottlenecks","date":"2021-09-29","arxiv_id":"2109.14320","repositories_listed":0,"syntology":null},{"url":null,"slug":"learnability-of-convolutional-neural-networks","title":"Learnability of convolutional neural networks for infinite dimensional input via mixed and anisotropic smoothness","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mlp-based-architecture-with-variable-length","title":"MLP-based architecture with variable length input for automatic speech recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"phasefool-phase-oriented-audio-adversarial","title":"PhaseFool: Phase-oriented Audio Adversarial Examples via Energy Dissipation","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-mlp-a-simple-mlp-architecture-for","title":"Speech-MLP: a simple MLP architecture for speech processing","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesising-audio-adversarial-examples-for","title":"Synthesising Audio Adversarial Examples for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-role-of-self-attention-for","title":"Understanding the Role of Self Attention for Efficient Speech Recognition","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"w-ctc-a-connectionist-temporal-classification","title":"W-CTC: a Connectionist Temporal Classification Loss with Wild Cards","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"will-a-blind-model-hear-better-advanced","title":"Will a Blind Model Hear Better? Advanced Audiovisual Recognition System with Brain-Like Compensating and Gating","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-dependency-coding-inspired-multimodal","title":"Neural Dependency Coding inspired Multimodal Fusion","date":"2021-09-28","arxiv_id":"2110.00385","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-language-model-adaptation-for-speech","title":"Private Language Model Adaptation for Speech Recognition","date":"2021-09-28","arxiv_id":"2110.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-level-confidence-estimation-for-rnn","title":"Word-level confidence estimation for RNN transducers","date":"2021-09-28","arxiv_id":"2110.15222","repositories_listed":0,"syntology":null},{"url":"/paper/bigssl-exploring-the-frontier-of-large-scale","slug":"bigssl-exploring-the-frontier-of-large-scale","title":"BigSSL: Exploring the Frontier of Large-Scale Semi-Supervised Learning for Automatic Speech Recognition","date":"2021-09-27","arxiv_id":"2109.13226","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenges-and-opportunities-of-speech","title":"Challenges and Opportunities of Speech Recognition for Bengali Language","date":"2021-09-27","arxiv_id":"2109.13217","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-model-robustness-to-automatic-speech","title":"Topic Model Robustness to Automatic Speech Recognition Errors in Podcast Transcripts","date":"2021-09-25","arxiv_id":"2109.12306","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimized-power-normalized-cepstral","title":"Optimized Power Normalized Cepstral Coefficients towards Robust Deep Speaker Verification","date":"2021-09-24","arxiv_id":"2109.12058","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-dynamic-filter-for-keyword","title":"A Lightweight dynamic filter for keyword spotting","date":"2021-09-23","arxiv_id":"2109.11165","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenario-aware-speech-recognition","title":"Scenario Aware Speech Recognition: Advancements for Apollo Fearless Steps & CHiME-4 Corpora","date":"2021-09-23","arxiv_id":"2109.11086","repositories_listed":0,"syntology":null},{"url":null,"slug":"animal-inspired-application-of-a-variant-of","title":"Animal inspired Application of a Variant of Mel Spectrogram for Seismic Data Processing","date":"2021-09-22","arxiv_id":"2109.10733","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-domain-specific-language-models-for","title":"Learning Domain Specific Language Models for Automatic Speech Recognition through Machine Translation","date":"2021-09-21","arxiv_id":"2110.10261","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-difficulty-of-segmenting-words-with","title":"On the Difficulty of Segmenting Words with Attention","date":"2021-09-21","arxiv_id":"2109.10107","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-recognition-is-worth-32","title":"Audio-Visual Speech Recognition is Worth 32$\\times$32$\\times$8 Voxels","date":"2021-09-20","arxiv_id":"2109.09536","repositories_listed":0,"syntology":null},{"url":null,"slug":"irnn-integer-only-recurrent-neural-network","title":"iRNN: Integer-only Recurrent Neural Network","date":"2021-09-20","arxiv_id":"2109.09828","repositories_listed":0,"syntology":null},{"url":null,"slug":"meetdot-videoconferencing-with-live","title":"MeetDot: Videoconferencing with Live Translation Captions","date":"2021-09-20","arxiv_id":"2109.09577","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-approach-for-measuring-the","title":"Model-Based Approach for Measuring the Fairness in ASR","date":"2021-09-19","arxiv_id":"2109.09061","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav-bert-cooperative-acoustic-and-linguistic","title":"Wav-BERT: Cooperative Acoustic and Linguistic Representation Learning for Low-Resource Speech Recognition","date":"2021-09-19","arxiv_id":"2109.09161","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-encoder-architecture-with-encoder","title":"Dual-Encoder Architecture with Encoder Selection for Joint Close-Talk and Far-Talk Speech Recognition","date":"2021-09-17","arxiv_id":"2109.08744","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-audio-textual-architecture-for","title":"Multimodal Audio-textual Architecture for Robust Spoken Language Understanding","date":"2021-09-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pdaugment-data-augmentation-by-pitch-and","title":"PDAugment: Data Augmentation by Pitch and Duration Adjustments for Automatic Lyrics Transcription","date":"2021-09-16","arxiv_id":"2109.07940","repositories_listed":0,"syntology":null},{"url":null,"slug":"utterance-level-neural-confidence-measure-for","title":"Utterance-level neural confidence measure for end-to-end children speech recognition","date":"2021-09-16","arxiv_id":"2109.07750","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accent-identification-and-accented","title":"Improving Accent Identification and Accented Speech Recognition Under a Framework of Self-supervised Learning","date":"2021-09-15","arxiv_id":"2109.07349","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-streaming-transformer-based-asr","title":"Improving Streaming Transformer Based ASR Under a Framework of Self-supervised Learning","date":"2021-09-15","arxiv_id":"2109.07327","repositories_listed":0,"syntology":null},{"url":null,"slug":"lrwr-large-scale-benchmark-for-lip-reading-in","title":"LRWR: Large-Scale Benchmark for Lip Reading in Russian language","date":"2021-09-14","arxiv_id":"2109.06692","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-transformer-with-unified","title":"Non-autoregressive Transformer with Unified Bidirectional Decoder for Automatic Speech Recognition","date":"2021-09-14","arxiv_id":"2109.06684","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-adapters-for-parameter-efficient-asr","title":"Residual Adapters for Parameter-Efficient ASR Adaptation to Atypical and Accented Speech","date":"2021-09-14","arxiv_id":"2109.06952","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-recurrent-neural-network-for","title":"Applications of Recurrent Neural Network for Biometric Authentication & Anomaly Detection","date":"2021-09-13","arxiv_id":"2109.05701","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-decidability-based-loss-function","title":"A Decidability-Based Loss Function","date":"2021-09-12","arxiv_id":"2109.05524","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-schemes-for","title":"Unsupervised Domain Adaptation Schemes for Building ASR in Low-resource Languages","date":"2021-09-12","arxiv_id":"2109.05494","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-vocabulary-audio-visual-speech","title":"Large-vocabulary Audio-visual Speech Recognition in Noisy Environments","date":"2021-09-10","arxiv_id":"2109.04894","repositories_listed":0,"syntology":null},{"url":null,"slug":"remember-the-context-asr-slot-error","title":"Remember the context! ASR slot error correction through memorization","date":"2021-09-10","arxiv_id":"2109.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-attention-channel-combinator-frontend","title":"Self-Attention Channel Combinator Frontend for End-to-End Multichannel Far-field Speech Recognition","date":"2021-09-10","arxiv_id":"2109.04783","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepemo-deep-learning-for-speech-emotion","title":"DeepEMO: Deep Learning for Speech Emotion Recognition","date":"2021-09-09","arxiv_id":"2109.04081","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-brief-history-of-ai-how-to-prevent-another","title":"A brief history of AI: how to prevent another winter (a critical review)","date":"2021-09-03","arxiv_id":"2109.01517","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-topological-framework-for-the-design-of","title":"Using Topological Framework for the Design of Activation Function and Model Pruning in Deep Neural Networks","date":"2021-09-03","arxiv_id":"2109.01572","repositories_listed":0,"syntology":null},{"url":null,"slug":"coarse-to-fine-and-cross-lingual-asr-transfer","title":"Coarse-To-Fine And Cross-Lingual ASR Transfer","date":"2021-09-02","arxiv_id":"2109.00916","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-of-end-to-end-automatic-speech-1","title":"Robustness of end-to-end Automatic Speech Recognition Models – A Case Study using Mozilla DeepSpeech","date":"2021-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-constrained-pointer-generator-for-end-to","title":"Tree-constrained Pointer Generator for End-to-end Contextual Speech Recognition","date":"2021-09-01","arxiv_id":"2109.00627","repositories_listed":0,"syntology":null},{"url":"/paper/asr-glue-a-new-multi-task-benchmark-for-asr","slug":"asr-glue-a-new-multi-task-benchmark-for-asr","title":"ASR-GLUE: A New Multi-task Benchmark for ASR-Robust Natural Language Understanding","date":"2021-08-30","arxiv_id":"2108.13048","repositories_listed":0,"syntology":null},{"url":"/paper/europarl-asr-a-large-corpus-of-parliamentary","slug":"europarl-asr-a-large-corpus-of-parliamentary","title":"Europarl-ASR: A Large Corpus of Parliamentary Debates for Streaming ASR Benchmarking and Speech Data Filtering/Verbatimization","date":"2021-08-30","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-transformer-transducer-for","title":"Multi-Channel Transformer Transducer for Speech Recognition","date":"2021-08-30","arxiv_id":"2108.12953","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-framework-for-video-ads","title":"A Multimodal Framework for Video Ads Understanding","date":"2021-08-29","arxiv_id":"2108.12868","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigations-on-speech-recognition-systems","title":"Investigations on Speech Recognition Systems for Low-Resource Dialectal Arabic-English Code-Switching Speech","date":"2021-08-29","arxiv_id":"2108.12881","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-driven-text-descriptions-for-images","title":"Goal-driven text descriptions for images","date":"2021-08-28","arxiv_id":"2108.12575","repositories_listed":0,"syntology":null},{"url":null,"slug":"4-bit-quantization-of-lstm-based-speech","title":"4-bit Quantization of LSTM-based Speech Recognition Models","date":"2021-08-27","arxiv_id":"2108.12074","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-retraining-free-speech-recognition","title":"Exploring Retraining-Free Speech Recognition for Intra-sentential Code-Switching","date":"2021-08-27","arxiv_id":"2109.00921","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-attention-bidirectional-deep-learning","title":"Full Attention Bidirectional Deep Learning Structure for Single Channel Speech Enhancement","date":"2021-08-27","arxiv_id":"2108.12105","repositories_listed":0,"syntology":null},{"url":null,"slug":"grammar-based-identification-of-speaker-role","title":"Grammar Based Speaker Role Identification for Air Traffic Control Speech Recognition","date":"2021-08-27","arxiv_id":"2108.12175","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-callsign-recognition-with-air","title":"Improving callsign recognition with air-surveillance data in air-traffic communication","date":"2021-08-27","arxiv_id":"2108.12156","repositories_listed":0,"syntology":null},{"url":null,"slug":"injecting-text-in-self-supervised-speech","title":"Injecting Text in Self-Supervised Speech Pretraining","date":"2021-08-27","arxiv_id":"2108.12226","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-aware-warping-factors-in-mask-based","title":"Task-aware Warping Factors in Mask-based Speech Enhancement","date":"2021-08-27","arxiv_id":"2108.12128","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-single-channel-speech","title":"Cross-domain Single-channel Speech Enhancement Model with Bi-projection Fusion Module for Noise-robust ASR","date":"2021-08-26","arxiv_id":"2108.11598","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-invariant-truecasing-with-a-word-and","title":"Position-Invariant Truecasing with a Word-and-Character Hierarchical Recurrent Neural Network","date":"2021-08-26","arxiv_id":"2108.11943","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-neural-networks-methods-applications","title":"Graph Neural Networks: Methods, Applications, and Opportunities","date":"2021-08-24","arxiv_id":"2108.10733","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-exposure-bias-in-training-recurrent","title":"Reducing Exposure Bias in Training Recurrent Neural Network Transducers","date":"2021-08-24","arxiv_id":"2108.10803","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-transformer-based-framework-for","title":"A Unified Transformer-based Framework for Duplex Text Normalization","date":"2021-08-23","arxiv_id":"2108.09889","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-using-limited","title":"Automatic Speech Recognition And Limited Vocabulary: A Survey","date":"2021-08-23","arxiv_id":"2108.10254","repositories_listed":0,"syntology":null},{"url":null,"slug":"subject-envelope-based-multitype","title":"Subject Envelope based Multitype Reconstruction Algorithm of Speech Samples of Parkinson's Disease","date":"2021-08-23","arxiv_id":"2108.09922","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-decoder-conformer-for-multilingual","title":"A Dual-Decoder Conformer for Multilingual Speech Recognition","date":"2021-08-22","arxiv_id":"2109.03277","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizing-rnn-transducer-to-out-domain","title":"Generalizing RNN-Transducer to Out-Domain Audio via Sparse Self-Attention Layers","date":"2021-08-22","arxiv_id":"2108.10752","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-speech-recognition-for-low","title":"Multilingual Speech Recognition for Low-Resource Indian Languages using Multi-Task conformer","date":"2021-08-22","arxiv_id":"2109.03969","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-summarization-for-longform","title":"Hierarchical Summarization for Longform Spoken Dialog","date":"2021-08-21","arxiv_id":"2108.09597","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-streams-and-two-resolution-spectrograms","title":"A Multi-level Acoustic Feature Extraction Framework for Transformer Based End-to-End Speech Recognition","date":"2021-08-18","arxiv_id":"2108.07980","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-light-weight-contextual-spelling-correction","title":"A Light-weight contextual spelling correction model for customizing transducer-based speech recognition systems","date":"2021-08-17","arxiv_id":"2108.07493","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexter-deep-encoding-of-external-knowledge","title":"DEXTER: Deep Encoding of External Knowledge for Named Entity Recognition in Virtual Assistants","date":"2021-08-15","arxiv_id":"2108.06633","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-training-set-selection-for-asr","title":"Multilingual training set selection for ASR in under-resourced Malian languages","date":"2021-08-13","arxiv_id":"2108.06164","repositories_listed":0,"syntology":null},{"url":null,"slug":"dereverberation-of-autoregressive-envelopes","title":"Dereverberation of Autoregressive Envelopes for Far-field Speech Recognition","date":"2021-08-12","arxiv_id":"2108.05520","repositories_listed":0,"syntology":null},{"url":null,"slug":"stargan-vc-asr-stargan-based-non-parallel","title":"StarGAN-VC+ASR: StarGAN-based Non-Parallel Voice Conversion Regularized by Automatic Speech Recognition","date":"2021-08-10","arxiv_id":"2108.04395","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hw-tsc-s-offline-speech-translation","title":"The HW-TSC's Offline Speech Translation Systems for IWSLT 2021 Evaluation","date":"2021-08-09","arxiv_id":"2108.03845","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-frequency-localization-using-deep","title":"Time-Frequency Localization Using Deep Convolutional Maxout Neural Network in Persian Speech Recognition","date":"2021-08-09","arxiv_id":"2108.03818","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-assessment-of-deep-learning","title":"An empirical assessment of deep learning approaches to task-oriented dialog management","date":"2021-08-07","arxiv_id":"2108.03478","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-attention-mechanism-and","title":"Spatio-Temporal Attention Mechanism and Knowledge Distillation for Lip Reading","date":"2021-08-07","arxiv_id":"2108.03543","repositories_listed":0,"syntology":null},{"url":null,"slug":"out-of-domain-generalization-from-a-single","title":"Out-of-Domain Generalization from a Single Source: An Uncertainty Quantification Approach","date":"2021-08-05","arxiv_id":"2108.02888","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-and-neural-network-guided-convolutional","title":"Blind and neural network-guided convolutional beamformer for joint denoising, dereverberation, and source separation","date":"2021-08-04","arxiv_id":"2108.01836","repositories_listed":0,"syntology":null},{"url":null,"slug":"dyn-asr-compact-multilingual-speech","title":"Dyn-ASR: Compact, Multilingual Speech Recognition via Spoken Language and Accent Identification","date":"2021-08-04","arxiv_id":"2108.02034","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-frequency-modulation-is-encoded","title":"Fast frequency modulation is encoded according to the listener expectations in the human subcortical auditory pathway","date":"2021-08-04","arxiv_id":"2108.02066","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-distinction-between-asr-errors-and","title":"Improving Distinction between ASR Errors and Speech Disfluencies with Feature Space Interpolation","date":"2021-08-04","arxiv_id":"2108.01812","repositories_listed":0,"syntology":null},{"url":null,"slug":"spartus-a-9-4-top-s-fpga-based-lstm","title":"Spartus: A 9.4 TOp/s FPGA-based LSTM Accelerator Exploiting Spatio-Temporal Sparsity","date":"2021-08-04","arxiv_id":"2108.02297","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-in-speech","title":"Unsupervised Domain Adaptation in Speech Recognition using Phonetic Features","date":"2021-08-04","arxiv_id":"2108.02850","repositories_listed":0,"syntology":null}],"record_sha256":"c8c37726fb6db22e3decb891751efbf22ea4f076bf4790b54a168b0bc31887a3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}