{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/40","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":40,"pages_in_order":58,"rows_per_page":100,"rows":[3901,4000],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/39","next":"/task/speech-recognition-1/papers/41","papers":[{"url":null,"slug":"leveraging-acoustic-and-linguistic-embeddings","title":"Leveraging Acoustic and Linguistic Embeddings from Pretrained speech and language Models for Intent Classification","date":"2021-02-15","arxiv_id":"2102.07370","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalization-strategies-for-end-to-end","title":"Personalization Strategies for End-to-End Speech Recognition Systems","date":"2021-02-15","arxiv_id":"2102.07739","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-classification-using-hidden-markov","title":"Robust Classification using Hidden Markov Models and Mixtures of Normalizing Flows","date":"2021-02-15","arxiv_id":"2102.07284","repositories_listed":0,"syntology":null},{"url":null,"slug":"thank-you-for-attention-a-survey-on-attention","title":"Thank you for Attention: A survey on Attention-based Artificial Neural Networks for Automatic Speech Recognition","date":"2021-02-14","arxiv_id":"2102.07259","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-apc-bidirectional-autoregressive","title":"Bi-APC: Bidirectional Autoregressive Predictive Coding for Unsupervised Pre-training and Its Application to Children's ASR","date":"2021-02-12","arxiv_id":"2102.06816","repositories_listed":0,"syntology":null},{"url":null,"slug":"content-aware-speaker-embeddings-for-speaker","title":"Content-Aware Speaker Embeddings for Speaker Diarisation","date":"2021-02-12","arxiv_id":"2102.06467","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-as-i-mean-not-as-i-say-sequence-loss","title":"Do as I mean, not as I say: Sequence Loss Training for Spoken Language Understanding","date":"2021-02-12","arxiv_id":"2102.06750","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-punctuation-prediction-with","title":"Multimodal Punctuation Prediction with Contextual Dropout","date":"2021-02-12","arxiv_id":"2102.11012","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-classification-learning-with-neural","title":"Fast Classification Learning with Neural Networks and Conceptors for Speech Recognition and Car Driving Maneuvers","date":"2021-02-10","arxiv_id":"2102.05588","repositories_listed":0,"syntology":null},{"url":null,"slug":"fused-acoustic-and-text-encoding-for","title":"Fused Acoustic and Text Encoding for Multimodal Bilingual Pretraining and Speech Translation","date":"2021-02-10","arxiv_id":"2102.05766","repositories_listed":0,"syntology":null},{"url":null,"slug":"nuva-a-naming-utterance-verifier-for-aphasia","title":"NUVA: A Naming Utterance Verifier for Aphasia Treatment","date":"2021-02-10","arxiv_id":"2102.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-transformer-language-models-for","title":"Bayesian Transformer Language Models for Speech Recognition","date":"2021-02-09","arxiv_id":"2102.04754","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsification-via-compressed-sensing-for","title":"Sparsification via Compressed Sensing for Automatic Speech Recognition","date":"2021-02-09","arxiv_id":"2102.04932","repositories_listed":0,"syntology":null},{"url":null,"slug":"train-your-classifier-first-cascade-neural","title":"Train your classifier first: Cascade Neural Networks Training from upper layers to lower layers","date":"2021-02-09","arxiv_id":"2102.04697","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-multi-channel-transformer-for","title":"End-to-End Multi-Channel Transformer for Speech Recognition","date":"2021-02-08","arxiv_id":"2102.03951","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-domain-speech-extraction-with-spatial","title":"Time-Domain Speech Extraction with Spatial Information and Multi Speaker Conditioning Mechanism","date":"2021-02-07","arxiv_id":"2102.03762","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bandit-approach-to-curriculum-generation","title":"A bandit approach to curriculum generation for automatic speech recognition","date":"2021-02-06","arxiv_id":"2102.03662","repositories_listed":0,"syntology":null},{"url":null,"slug":"intermediate-loss-regularization-for-ctc","title":"Intermediate Loss Regularization for CTC-based Speech Recognition","date":"2021-02-05","arxiv_id":"2102.03216","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-self-supervised-pre-training-for","title":"Multi-Task Self-Supervised Pre-Training for Music Classification","date":"2021-02-05","arxiv_id":"2102.03229","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-augmentation-and-adaptive-ctc","title":"Two-Stage Augmentation and Adaptive CTC Fusion for Improved Robustness of Multi-Stream End-to-End ASR","date":"2021-02-05","arxiv_id":"2102.03055","repositories_listed":0,"syntology":null},{"url":null,"slug":"effects-of-number-of-filters-of-convolutional","title":"Effects of Number of Filters of Convolutional Layers on Speech Recognition Model Accuracy","date":"2021-02-03","arxiv_id":"2102.02326","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-purpose-speech-representation","title":"General-Purpose Speech Representation Learning through a Self-Supervised Multi-Granularity Framework","date":"2021-02-03","arxiv_id":"2102.01930","repositories_listed":0,"syntology":null},{"url":null,"slug":"internal-language-model-training-for-domain","title":"Internal Language Model Training for Domain-Adaptive End-to-End Speech Recognition","date":"2021-02-02","arxiv_id":"2102.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-multilingual-tedx-corpus-for-speech","title":"The Multilingual TEDx Corpus for Speech Recognition and Translation","date":"2021-02-02","arxiv_id":"2102.01757","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-scaling-contrastive-representations-for","title":"On Scaling Contrastive Representations for Low-Resource Speech Recognition","date":"2021-02-01","arxiv_id":"2102.00850","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-attack-detection-approach-for-iiot","title":"Robust Attack Detection Approach for IIoT Using Ensemble Classifier","date":"2021-01-30","arxiv_id":"2102.01515","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-by-simply-fine-tuning-bert","title":"Speech Recognition by Simply Fine-tuning BERT","date":"2021-01-30","arxiv_id":"2102.00291","repositories_listed":0,"syntology":null},{"url":null,"slug":"bcn2brno-asr-system-fusion-for-albayzin-2020","title":"BCN2BRNO: ASR System Fusion for Albayzin 2020 Speech to Text Challenge","date":"2021-01-29","arxiv_id":"2101.12729","repositories_listed":0,"syntology":null},{"url":null,"slug":"bounds-on-mutual-information-of-mixture-data","title":"Bounds on mutual information of mixture data for classification tasks","date":"2021-01-27","arxiv_id":"2101.11670","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-deliberation-for-two-pass","title":"Transformer Based Deliberation for Two-Pass Speech Recognition","date":"2021-01-27","arxiv_id":"2101.11577","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-end-to-end-asr-for-endangered","title":"Leveraging End-to-End ASR for Endangered Language Documentation: An Empirical Study on Yoloxóchitl Mixtec","date":"2021-01-26","arxiv_id":"2101.10877","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-speaker-diarization-recent","title":"A Review of Speaker Diarization: Recent Advances with Deep Learning","date":"2021-01-24","arxiv_id":"2101.09624","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-beam-search-confidence-for-energy","title":"Exploiting Beam Search Confidence for Energy-Efficient Speech Recognition","date":"2021-01-22","arxiv_id":"2101.09083","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-models-for-joint-speech-recognition","title":"Streaming Models for Joint Speech Recognition and Translation","date":"2021-01-22","arxiv_id":"2101.09149","repositories_listed":0,"syntology":null},{"url":null,"slug":"noisy-target-training-a-training-strategy-for","title":"Noisy-target Training: A Training Strategy for DNN-based Speech Enhancement without Clean Speech","date":"2021-01-21","arxiv_id":"2101.08625","repositories_listed":0,"syntology":null},{"url":null,"slug":"vote400-voide-of-the-elderly-400-hours-a","title":"VOTE400(Voide Of The Elderly 400 Hours): A Speech Dataset to Study Voice Interface for Elderly-Care","date":"2021-01-20","arxiv_id":"2101.11469","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-wav2vec2-0-and-bert-into-end-to-end","title":"Efficiently Fusing Pretrained Acoustic and Linguistic Encoders for Low-resource Speech Recognition","date":"2021-01-17","arxiv_id":"2101.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evaluation-of-word-level-confidence","title":"An evaluation of word-level confidence estimation for end-to-end automatic speech recognition","date":"2021-01-14","arxiv_id":"2101.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-offline-transformer-based-end-to-end","title":"Fast offline Transformer-based end-to-end automatic speech recognition for real-world applications","date":"2021-01-14","arxiv_id":"2101.05600","repositories_listed":0,"syntology":null},{"url":null,"slug":"wer-bert-automatic-wer-estimation-with-bert","title":"WER-BERT: Automatic WER Estimation with BERT in a Balanced Ordinal Classification Paradigm","date":"2021-01-14","arxiv_id":"2101.05478","repositories_listed":0,"syntology":null},{"url":null,"slug":"brds-an-fpga-based-lstm-accelerator-with-row","title":"BRDS: An FPGA-based LSTM Accelerator with Row-Balanced Dual-Ratio Sparsification","date":"2021-01-07","arxiv_id":"2101.02667","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypothesis-stitcher-for-end-to-end-speaker","title":"Hypothesis Stitcher for End-to-End Speaker-attributed ASR on Long-form Multi-talker Recordings","date":"2021-01-06","arxiv_id":"2101.01853","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-aware-neural-language-models-for","title":"Domain-aware Neural Language Models for Speech Recognition","date":"2021-01-05","arxiv_id":"2101.03229","repositories_listed":0,"syntology":null},{"url":null,"slug":"hmm-based-phoneme-speech-recognition-system-1","title":"HMM-based phoneme speech recognition system for the control and command of industrial robots","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-monotonic-alignments-with-source","title":"Learning Monotonic Alignments with Source-Aware GMM Attention","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-without-forgetting-task-aware","title":"Learning without Forgetting: Task Aware Multitask Learning for Multi-Modality Tasks","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nas-bench-asr-reproducible-neural","title":"NAS-Bench-ASR: Reproducible Neural Architecture Search for Speech Recognition","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsifying-networks-via-subdifferential","title":"Sparsifying Networks via Subdifferential Inclusion","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-simpler-the-better-vanilla-sgd-revisited","title":"The simpler the better: vanilla sgd revisited","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertain-out-of-domain-generalization","title":"Uncertain Out-of-Domain Generalization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"why-does-decentralized-training-outperform","title":"Why Does Decentralized Training Outperform Synchronous Training In The Large Batch Setting?","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-multi-frame-adl-mvdr-for-target","title":"Multi-channel Multi-frame ADL-MVDR for Target Speech Separation","date":"2020-12-24","arxiv_id":"2012.13442","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-as-augmentation-for-low","title":"Speech Synthesis as Augmentation for Low-Resource ASR","date":"2020-12-23","arxiv_id":"2012.13004","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-reasoning-graph-neural-network","title":"A Hierarchical Reasoning Graph Neural Network for The Automatic Scoring of Answer Transcriptions in Video Job Interviews","date":"2020-12-22","arxiv_id":"2012.11960","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-meta-sampling-for-multilingual","title":"Adversarial Meta Sampling for Multilingual Low-Resource Speech Recognition","date":"2020-12-22","arxiv_id":"2012.11896","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-wav2vec2-0-to-speech-recognition-in","title":"Applying Wav2vec2.0 to Speech Recognition in Various Low-resource Languages","date":"2020-12-22","arxiv_id":"2012.12121","repositories_listed":0,"syntology":null},{"url":null,"slug":"limitations-of-deep-neural-networks-a","title":"Limitations of Deep Neural Networks: a discussion of G. Marcus' critical appraisal of deep learning","date":"2020-12-22","arxiv_id":"2012.15754","repositories_listed":0,"syntology":null},{"url":null,"slug":"adjust-free-adversarial-example-generation-in","title":"Adjust-free adversarial example generation in speech recognition using evolutionary multi-objective optimization under black-box condition","date":"2020-12-21","arxiv_id":"2012.11138","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-methods-for-effective-efficient-and","title":"Neural Methods for Effective, Efficient, and Exposure-Aware Information Retrieval","date":"2020-12-21","arxiv_id":"2012.11685","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-streaming-asr-with-non-autoregressive","title":"Toward Streaming ASR with Non-Autoregressive Insertion-based Model","date":"2020-12-18","arxiv_id":"2012.10128","repositories_listed":0,"syntology":null},{"url":null,"slug":"cif-based-collaborative-decoding-for-end-to","title":"CIF-based Collaborative Decoding for End-to-end Contextual Speech Recognition","date":"2020-12-17","arxiv_id":"2012.09466","repositories_listed":0,"syntology":null},{"url":"/paper/exploring-transfer-learning-for-end-to-end","slug":"exploring-transfer-learning-for-end-to-end","title":"Exploring Transfer Learning For End-to-End Spoken Language Understanding","date":"2020-12-15","arxiv_id":"2012.08549","repositories_listed":0,"syntology":null},{"url":null,"slug":"user-friendly-automatic-transcription-of-low","title":"User-friendly automatic transcription of low-resource languages: Plugging ESPnet into Elpis","date":"2020-12-15","arxiv_id":"2101.03027","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-on-device-fully-neural-end-to-end","title":"A review of on-device fully neural end-to-end automatic speech recognition algorithms","date":"2020-12-14","arxiv_id":"2012.07974","repositories_listed":0,"syntology":null},{"url":"/paper/part-based-lipreading-for-audio-visual-speech","slug":"part-based-lipreading-for-audio-visual-speech","title":"Part-based Lipreading for Audio-Visual Speech Recognition","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"less-is-more-improved-rnn-t-decoding-using","title":"Less Is More: Improved RNN-T Decoding Using Limited Label Context and Path Merging","date":"2020-12-12","arxiv_id":"2012.06749","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-wav2vec-2-0-on-speaker-verification","title":"Exploring wav2vec 2.0 on speaker verification and language identification","date":"2020-12-11","arxiv_id":"2012.06185","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-robustness-to-disfluencies-in-rnn","title":"Improved Robustness to Disfluencies in RNN-Transducer Based Speech Recognition","date":"2020-12-11","arxiv_id":"2012.06259","repositories_listed":0,"syntology":null},{"url":null,"slug":"next-wave-artificial-intelligence-robust","title":"Next Wave Artificial Intelligence: Robust, Explainable, Adaptable, Ethical, and Accountable","date":"2020-12-11","arxiv_id":"2012.06058","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-classifier-interactive-learning-for","title":"Multi-Classifier Interactive Learning for Ambiguous Speech Emotion Recognition","date":"2020-12-10","arxiv_id":"2012.05429","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-for-endangered-and-extinct","title":"Speech Recognition for Endangered and Extinct Samoyedic languages","date":"2020-12-09","arxiv_id":"2012.05331","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-learning-of-lf-mmi-trained-time","title":"Bayesian Learning of LF-MMI Trained Time Delay Neural Networks for Speech Recognition","date":"2020-12-08","arxiv_id":"2012.04494","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-multiple-asr-hypotheses-to-boost-i18n","title":"Using multiple ASR hypotheses to boost i18n NLU performance","date":"2020-12-07","arxiv_id":"2012.04099","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapt-and-adjust-overcoming-the-long-tail-1","title":"Adapt-and-Adjust: Overcoming the Long-Tail Problem of Multilingual Speech Recognition","date":"2020-12-03","arxiv_id":"2012.01687","repositories_listed":0,"syntology":null},{"url":"/paper/100000-podcasts-a-spoken-english-document","slug":"100000-podcasts-a-spoken-english-document","title":"100,000 Podcasts: A Spoken English Document Corpus","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-for-non-standardised-languages-with","title":"ASR for Non-standardised Languages with Dialectal Variation: the case of Swiss German","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-hybrid-algorithm-for-automatic","title":"Development of Hybrid Algorithm for Automatic Extraction of Multiword Expressions from Monolingual and Parallel Corpus of English and Punjabi","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-marginal-personalization-for-asr","title":"Federated Marginal Personalization for ASR Rescoring","date":"2020-12-01","arxiv_id":"2012.00898","repositories_listed":0,"syntology":null},{"url":null,"slug":"german-arabic-speech-to-speech-translation","title":"German-Arabic Speech-to-Speech Translation for Psychiatric Diagnosis","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-learning-of-spoken-language","title":"Multi-task Learning of Spoken Language Understanding by Integrating N-Best Hypotheses with Hierarchical Attention","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-detection-of-sentence-completion","title":"On-Device detection of sentence completion for voice assistants with low-memory footprint","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-transcription","title":"Sparse Transcription","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-adaptation-of-sequence-to-sequence","title":"Supervised Adaptation of Sequence-to-Sequence Speech Recognition Systems using Batch-Weighting","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-indigenous-languages-technology-project","title":"The Indigenous Languages Technology project at NRC Canada: An empowerment-oriented approach to developing language software","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accuracy-of-rare-words-for-rnn","title":"Improving accuracy of rare words for RNN-Transducer through unigram shallow fusion","date":"2020-11-30","arxiv_id":"2012.00133","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-detection-of-alzheimer-s-disease","title":"Multi-Modal Detection of Alzheimer's Disease from Speech and Text","date":"2020-11-30","arxiv_id":"2012.00096","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-transducers-for-code-switched","title":"Transformer-Transducers for Code-Switched Speech Recognition","date":"2020-11-30","arxiv_id":"2011.15023","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-end-to-end-multi-talker-speech","title":"Streaming end-to-end multi-talker speech recognition","date":"2020-11-26","arxiv_id":"2011.13148","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-domain-adaptation-for-speech","title":"Unsupervised Domain Adaptation for Speech Recognition via Uncertainty Driven Self-Training","date":"2020-11-26","arxiv_id":"2011.13439","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-panoramic-survey-of-natural-language","title":"A Panoramic Survey of Natural Language Processing in the Arab World","date":"2020-11-25","arxiv_id":"2011.12631","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrap-an-end-to-end-asr-system-by","title":"Bootstrap an end-to-end ASR system by multilingual training, transfer learning, text-to-text mapping and synthetic audio","date":"2020-11-25","arxiv_id":"2011.12696","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-recent-advances-of-binary-neural","title":"A Review of Recent Advances of Binary Neural Networks for Edge Computing","date":"2020-11-24","arxiv_id":"2011.14824","repositories_listed":0,"syntology":null},{"url":null,"slug":"adam-a-stochastic-method-with-adaptive-1","title":"Adam$^+$: A Stochastic Method with Adaptive Variance Reduction","date":"2020-11-24","arxiv_id":"2011.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"synth2aug-cross-domain-speaker-recognition","title":"Synth2Aug: Cross-domain speaker recognition with TTS synthesized speech","date":"2020-11-24","arxiv_id":"2011.11818","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-language-modeling-for-improving","title":"Multi-task Language Modeling for Improving Speech Recognition of Rare Words","date":"2020-11-23","arxiv_id":"2011.11715","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-multi-speaker-asr-with-rnn-t","title":"Streaming Multi-speaker ASR with RNN-T","date":"2020-11-23","arxiv_id":"2011.11671","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-synthetic-audio-to-improve-the","title":"Using Synthetic Audio to Improve The Recognition of Out-Of-Vocabulary Words in End-To-End ASR Systems","date":"2020-11-23","arxiv_id":"2011.11564","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-in-eeg-advance-of-the-last-ten","title":"Deep Learning in EEG: Advance of the Last Ten-Year Critical Period","date":"2020-11-22","arxiv_id":"2011.11128","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-t-asr-accuracy-using-context","title":"Improving RNN-T ASR Accuracy Using Context Audio","date":"2020-11-20","arxiv_id":"2011.10538","repositories_listed":0,"syntology":null},{"url":null,"slug":"tal-a-synchronised-multi-speaker-corpus-of","title":"TaL: a synchronised multi-speaker corpus of ultrasound tongue imaging, audio, and lip videos","date":"2020-11-19","arxiv_id":"2011.09804","repositories_listed":0,"syntology":null}],"record_sha256":"c7190619c4aba4c2fed1e736d07116960fbd3de0fc7436f197857dd73641f619","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}