{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/6","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":31,"rows_per_page":100,"rows":[501,600],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/5","next":"/task/automatic-speech-recognition/papers/7","papers":[{"url":"/paper/on-knowledge-distillation-for-direct-speech","slug":"on-knowledge-distillation-for-direct-speech","title":"On Knowledge Distillation for Direct Speech Translation","date":"2020-12-09","arxiv_id":"2012.04964","repositories_listed":1,"syntology":null},{"url":"/paper/mls-a-large-scale-multilingual-dataset-for","slug":"mls-a-large-scale-multilingual-dataset-for","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","date":"2020-12-07","arxiv_id":"2012.03411","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-asr-system-with-automatic","slug":"end-to-end-asr-system-with-automatic","title":"End to End ASR System with Automatic Punctuation Insertion","date":"2020-12-03","arxiv_id":"2012.02012","repositories_listed":1,"syntology":null},{"url":"/paper/attentively-embracing-noise-for-robust-latent","slug":"attentively-embracing-noise-for-robust-latent","title":"Attentively Embracing Noise for Robust Latent Representation in BERT","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-automatic-speech-recognition-for","slug":"end-to-end-automatic-speech-recognition-for","title":"End-to-End Automatic Speech Recognition for Gujarati","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/metacat-a-metadata-based-task-oriented","slug":"metacat-a-metadata-based-task-oriented","title":"metaCAT: A Metadata-based Task-oriented Chatbot Annotation Tool","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/wpd-an-improved-neural-beamformer-for","slug":"wpd-an-improved-neural-beamformer-for","title":"WPD++: An Improved Neural Beamformer for Simultaneous Speech Separation and Dereverberation","date":"2020-11-18","arxiv_id":"2011.09162","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-neural-architecture-search-for-end","slug":"efficient-neural-architecture-search-for-end","title":"Efficient Neural Architecture Search for End-to-end Speech Recognition via Straight-Through Gradients","date":"2020-11-11","arxiv_id":"2011.05649","repositories_listed":1,"syntology":null},{"url":"/paper/improving-rnn-transducer-based-asr-with","slug":"improving-rnn-transducer-based-asr-with","title":"Improving RNN Transducer Based ASR with Auxiliary Tasks","date":"2020-11-05","arxiv_id":"2011.03109","repositories_listed":1,"syntology":null},{"url":"/paper/minimum-bayes-risk-training-for-end-to-end","slug":"minimum-bayes-risk-training-for-end-to-end","title":"Minimum Bayes Risk Training for End-to-End Speaker-Attributed ASR","date":"2020-11-03","arxiv_id":"2011.02921","repositories_listed":1,"syntology":null},{"url":"/paper/dual-decoder-transformer-for-joint-automatic","slug":"dual-decoder-transformer-for-joint-automatic","title":"Dual-decoder Transformer for Joint Automatic Speech Recognition and Multilingual Speech Translation","date":"2020-11-02","arxiv_id":"2011.00747","repositories_listed":1,"syntology":null},{"url":"/paper/direct-segmentation-models-for-streaming","slug":"direct-segmentation-models-for-streaming","title":"Direct Segmentation Models for Streaming Speech Translation","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/punctuation-restoration-using-transformer","slug":"punctuation-restoration-using-transformer","title":"Punctuation Restoration using Transformer Models for High-and Low-Resource Languages","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-bottleneck-features-for","slug":"multilingual-bottleneck-features-for","title":"Multilingual Bottleneck Features for Improving ASR Performance of Code-Switched Speech in Under-Resourced Languages","date":"2020-10-31","arxiv_id":"2011.03118","repositories_listed":1,"syntology":null},{"url":"/paper/joint-masked-cpc-and-ctc-training-for-asr","slug":"joint-masked-cpc-and-ctc-training-for-asr","title":"Joint Masked CPC and CTC Training for ASR","date":"2020-10-30","arxiv_id":"2011.00093","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-end-to-end-multilingual-speech","slug":"large-scale-end-to-end-multilingual-speech","title":"Large-Scale End-to-End Multilingual Speech Recognition and Language Identification with Multi-Task Learning","date":"2020-10-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-estimation-for-attention-based","slug":"confidence-estimation-for-attention-based","title":"Confidence Estimation for Attention-based Sequence-to-sequence Models for Speech Recognition","date":"2020-10-22","arxiv_id":"2010.11428","repositories_listed":1,"syntology":null},{"url":"/paper/how-phonotactics-affect-multilingual-and-zero","slug":"how-phonotactics-affect-multilingual-and-zero","title":"How Phonotactics Affect Multilingual and Zero-shot ASR Performance","date":"2020-10-22","arxiv_id":"2010.12104","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-evaluation-in-asr-are-our-models","slug":"rethinking-evaluation-in-asr-are-our-models","title":"Rethinking Evaluation in ASR: Are Our Models Robust Enough?","date":"2020-10-22","arxiv_id":"2010.11745","repositories_listed":1,"syntology":null},{"url":"/paper/fastemit-low-latency-streaming-asr-with","slug":"fastemit-low-latency-streaming-asr-with","title":"FastEmit: Low-latency Streaming ASR with Sequence-level Emission Regularization","date":"2020-10-21","arxiv_id":"2010.11148","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/fastemit-low-latency-streaming-asr-with#ran","syntology_url":"https://syntology.ai/paper/2010.11148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11148"}},"official":null}},{"url":"/paper/towards-end-to-end-training-of-automatic","slug":"towards-end-to-end-training-of-automatic","title":"Towards End-to-End Training of Automatic Speech Recognition for Nigerian Pidgin","date":"2020-10-21","arxiv_id":"2010.11123","repositories_listed":1,"syntology":null},{"url":"/paper/venomave-clean-label-poisoning-against-speech","slug":"venomave-clean-label-poisoning-against-speech","title":"VenoMave: Targeted Poisoning Against Speech Recognition","date":"2020-10-21","arxiv_id":"2010.10682","repositories_listed":1,"syntology":null},{"url":"/paper/pushing-the-limits-of-semi-supervised","slug":"pushing-the-limits-of-semi-supervised","title":"Pushing the Limits of Semi-Supervised Learning for Automatic Speech Recognition","date":"2020-10-20","arxiv_id":"2010.10504","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pushing-the-limits-of-semi-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.10504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10504"}},"official":null}},{"url":"/paper/google-crowdsourced-speech-corpora-and","slug":"google-crowdsourced-speech-corpora-and","title":"Google Crowdsourced Speech Corpora and Related Open-Source Resources for Low-Resource Languages and Dialects: An Overview","date":"2020-10-14","arxiv_id":"2010.06778","repositories_listed":1,"syntology":null},{"url":"/paper/swiss-parliaments-corpus-an-automatically","slug":"swiss-parliaments-corpus-an-automatically","title":"Swiss Parliaments Corpus, an Automatically Aligned Swiss German Speech to Standard German Text Corpus","date":"2020-10-06","arxiv_id":"2010.02810","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-grounding-for-multimodal-speech","slug":"fine-grained-grounding-for-multimodal-speech","title":"Fine-Grained Grounding for Multimodal Speech Recognition","date":"2020-10-05","arxiv_id":"2010.02384","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-speech-2d-feature","slug":"end-to-end-learning-of-speech-2d-feature","title":"End-to-End Learning of Speech 2D Feature-Trajectory for Prosthetic Hands","date":"2020-09-22","arxiv_id":"2009.10283","repositories_listed":1,"syntology":null},{"url":"/paper/data-augmentation-using-prosody-and-false","slug":"data-augmentation-using-prosody-and-false","title":"Data augmentation using prosody and false starts to recognize non-native children's speech","date":"2020-08-29","arxiv_id":"2008.12914","repositories_listed":1,"syntology":null},{"url":"/paper/are-neural-open-domain-dialog-systems-robust","slug":"are-neural-open-domain-dialog-systems-robust","title":"Are Neural Open-Domain Dialog Systems Robust to Speech Recognition Errors in the Dialog History? An Empirical Study","date":"2020-08-18","arxiv_id":"2008.07683","repositories_listed":1,"syntology":null},{"url":"/paper/sum-product-networks-for-robust-automatic","slug":"sum-product-networks-for-robust-automatic","title":"Sum-Product Networks for Robust Automatic Speaker Identification","date":"2020-08-13","arxiv_id":"1910.11969","repositories_listed":1,"syntology":null},{"url":"/paper/investigation-of-end-to-end-speaker","slug":"investigation-of-end-to-end-speaker","title":"Investigation of End-To-End Speaker-Attributed ASR for Continuous Multi-Talker Recordings","date":"2020-08-11","arxiv_id":"2008.04546","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-the-knowledge-of-bert-for-sequence","slug":"distilling-the-knowledge-of-bert-for-sequence","title":"Distilling the Knowledge of BERT for Sequence-to-Sequence ASR","date":"2020-08-09","arxiv_id":"2008.03822","repositories_listed":1,"syntology":null},{"url":"/paper/word-error-rate-estimation-without-asr-output","slug":"word-error-rate-estimation-without-asr-output","title":"Word Error Rate Estimation Without ASR Output: e-WER2","date":"2020-08-08","arxiv_id":"2008.03403","repositories_listed":1,"syntology":null},{"url":"/paper/fast-transformers-with-clustered-attention","slug":"fast-transformers-with-clustered-attention","title":"Fast Transformers with Clustered Attention","date":"2020-07-09","arxiv_id":"2007.04825","repositories_listed":1,"syntology":null},{"url":"/paper/avlnet-learning-audio-visual-language","slug":"avlnet-learning-audio-visual-language","title":"AVLnet: Learning Audio-Visual Language Representations from Instructional Videos","date":"2020-06-16","arxiv_id":"2006.09199","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-neural-architectures-trained","slug":"evaluation-of-neural-architectures-trained","title":"Evaluation of Neural Architectures Trained with Square Loss vs Cross-Entropy in Classification Tasks","date":"2020-06-12","arxiv_id":"2006.07322","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-count-words-in-fluent-speech","slug":"learning-to-count-words-in-fluent-speech","title":"Learning to Count Words in Fluent Speech enables Online Speech Recognition","date":"2020-06-08","arxiv_id":"2006.04928","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-comparison-of-popular-end-to-end","slug":"on-the-comparison-of-popular-end-to-end","title":"On the Comparison of Popular End-to-End Models for Large Scale Speech Recognition","date":"2020-05-28","arxiv_id":"2005.14327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-comparison-of-popular-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2005.14327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14327"}},"official":{"repos":["cywang97/StreamingTransformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-end-to-end-speech-recognition-for","slug":"adapting-end-to-end-speech-recognition-for","title":"Adapting End-to-End Speech Recognition for Readable Subtitles","date":"2020-05-25","arxiv_id":"2005.12143","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-adversarial-examples-for-speech","slug":"detecting-adversarial-examples-for-speech","title":"Detecting Adversarial Examples for Speech Recognition via Uncertainty Quantification","date":"2020-05-24","arxiv_id":"2005.14611","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-named-entity-recognition-from","slug":"end-to-end-named-entity-recognition-from","title":"End-to-end Named Entity Recognition from English Speech","date":"2020-05-22","arxiv_id":"2005.11184","repositories_listed":1,"syntology":null},{"url":"/paper/pychain-a-fully-parallelized-pytorch","slug":"pychain-a-fully-parallelized-pytorch","title":"PyChain: A Fully Parallelized PyTorch Implementation of LF-MMI for End-to-End ASR","date":"2020-05-20","arxiv_id":"2005.09824","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-monotonic-multihead-attention-for","slug":"enhancing-monotonic-multihead-attention-for","title":"Enhancing Monotonic Multihead Attention for Streaming ASR","date":"2020-05-19","arxiv_id":"2005.09394","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/enhancing-monotonic-multihead-attention-for#ran","syntology_url":"https://syntology.ai/paper/2005.09394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.09394"}},"official":{"repos":["hirofumi0810/neural_sp"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-adversarial-training-data","slug":"generative-adversarial-training-data","title":"Generative Adversarial Training Data Adaptation for Very Low-resource Automatic Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09256","repositories_listed":1,"syntology":null},{"url":"/paper/improved-noisy-student-training-for-automatic","slug":"improved-noisy-student-training-for-automatic","title":"Improved Noisy Student Training for Automatic Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09629","repositories_listed":1,"syntology":null},{"url":"/paper/investigations-on-phoneme-based-end-to-end","slug":"investigations-on-phoneme-based-end-to-end","title":"A systematic comparison of grapheme-based vs. phoneme-based label units for encoder-decoder-attention models","date":"2020-05-19","arxiv_id":"2005.09336","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-pseudo-labeling-for-speech","slug":"iterative-pseudo-labeling-for-speech","title":"Iterative Pseudo-Labeling for Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09267","repositories_listed":1,"syntology":null},{"url":"/paper/quaternion-neural-networks-for-multi-channel","slug":"quaternion-neural-networks-for-multi-channel","title":"Quaternion Neural Networks for Multi-channel Distant Speech Recognition","date":"2020-05-18","arxiv_id":"2005.08566","repositories_listed":1,"syntology":null},{"url":"/paper/coupled-training-of-sequence-to-sequence","slug":"coupled-training-of-sequence-to-sequence","title":"Coupled Training of Sequence-to-Sequence Models for Accented Speech Recognition","date":"2020-05-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ctc-synchronous-training-for-monotonic","slug":"ctc-synchronous-training-for-monotonic","title":"CTC-synchronous Training for Monotonic Attention Model","date":"2020-05-10","arxiv_id":"2005.04712","repositories_listed":1,"syntology":null},{"url":"/paper/a-convolutional-neural-network-model-of-human","slug":"a-convolutional-neural-network-model-of-human","title":"A convolutional neural-network model of human cochlear mechanics and filter tuning for real-time applications","date":"2020-04-30","arxiv_id":"2004.14832","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-instructional-videos-probing-for-more","slug":"beyond-instructional-videos-probing-for-more","title":"Beyond Instructional Videos: Probing for More Diverse Visual-Textual Grounding on YouTube","date":"2020-04-29","arxiv_id":"2004.14338","repositories_listed":1,"syntology":null},{"url":"/paper/espnet-st-all-in-one-speech-translation","slug":"espnet-st-all-in-one-speech-translation","title":"ESPnet-ST: All-in-One Speech Translation Toolkit","date":"2020-04-21","arxiv_id":"2004.10234","repositories_listed":1,"syntology":null},{"url":"/paper/clovacall-korean-goal-oriented-dialog-speech","slug":"clovacall-korean-goal-oriented-dialog-speech","title":"ClovaCall: Korean Goal-Oriented Dialog Speech Corpus for Automatic Speech Recognition of Contact Centers","date":"2020-04-20","arxiv_id":"2004.09367","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-grapheme-to-phoneme-1","slug":"transformer-based-grapheme-to-phoneme-1","title":"Transformer based Grapheme-to-Phoneme Conversion","date":"2020-04-14","arxiv_id":"2004.06338","repositories_listed":1,"syntology":null},{"url":"/paper/morfessor-emprune-improved-subword","slug":"morfessor-emprune-improved-subword","title":"Morfessor EM+Prune: Improved Subword Segmentation with Expectation Maximization and Pruning","date":"2020-03-06","arxiv_id":"2003.03131","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-processing-advancements-by","slug":"natural-language-processing-advancements-by","title":"Natural Language Processing Advancements By Deep Learning: A Survey","date":"2020-03-02","arxiv_id":"2003.01200","repositories_listed":1,"syntology":null},{"url":"/paper/skinaugment-auto-encoding-speaker-conversions","slug":"skinaugment-auto-encoding-speaker-conversions","title":"SkinAugment: Auto-Encoding Speaker Conversions for Automatic Speech Translation","date":"2020-02-27","arxiv_id":"2002.12231","repositories_listed":1,"syntology":null},{"url":"/paper/attentional-speech-recognition-models","slug":"attentional-speech-recognition-models","title":"Attentional Speech Recognition Models Misbehave on Out-of-domain Utterances","date":"2020-02-12","arxiv_id":"2002.05150","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-speech-separation-dataset-and","slug":"continuous-speech-separation-dataset-and","title":"Continuous speech separation: dataset and analysis","date":"2020-01-30","arxiv_id":"2001.11482","repositories_listed":1,"syntology":null},{"url":"/paper/deep-xi-as-a-front-end-for-robust-automatic","slug":"deep-xi-as-a-front-end-for-robust-automatic","title":"Deep Xi as a Front-End for Robust Automatic Speech Recognition","date":"2020-01-28","arxiv_id":"1906.07319","repositories_listed":1,"syntology":null},{"url":"/paper/submodular-rank-aggregation-on-score-based","slug":"submodular-rank-aggregation-on-score-based","title":"Submodular Rank Aggregation on Score-based Permutations for Distributed Automatic Speech Recognition","date":"2020-01-27","arxiv_id":"2001.10529","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-labeling-approach-to-the-task-of","slug":"sequence-labeling-approach-to-the-task-of","title":"Sequence Labeling Approach to the Task of Sentence Boundary Detection","date":"2020-01-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/generating-synthetic-audio-data-for-attention","slug":"generating-synthetic-audio-data-for-attention","title":"Generating Synthetic Audio Data for Attention-Based Speech Recognition Systems","date":"2019-12-19","arxiv_id":"1912.09257","repositories_listed":1,"syntology":null},{"url":"/paper/synchronous-speech-recognition-and-speech-to","slug":"synchronous-speech-recognition-and-speech-to","title":"Synchronous Speech Recognition and Speech-to-Text Translation with Interactive Decoding","date":"2019-12-16","arxiv_id":"1912.07240","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-mask-for-transformer-based-end-to","slug":"semantic-mask-for-transformer-based-end-to","title":"Semantic Mask for Transformer based End-to-End Speech Recognition","date":"2019-12-06","arxiv_id":"1912.03010","repositories_listed":1,"syntology":null},{"url":"/paper/deep-contextualized-acoustic-representations","slug":"deep-contextualized-acoustic-representations","title":"Deep Contextualized Acoustic Representations For Semi-Supervised Speech Recognition","date":"2019-12-03","arxiv_id":"1912.01679","repositories_listed":1,"syntology":null},{"url":"/paper/improving-voice-separation-by-incorporating","slug":"improving-voice-separation-by-incorporating","title":"Improving Voice Separation by Incorporating End-to-end Speech Recognition","date":"2019-11-29","arxiv_id":"1911.12928","repositories_listed":1,"syntology":null},{"url":"/paper/kurdish-sorani-speech-to-text-presenting-an","slug":"kurdish-sorani-speech-to-text-presenting-an","title":"Kurdish (Sorani) Speech to Text: Presenting an Experimental Dataset","date":"2019-11-29","arxiv_id":"1911.13087","repositories_listed":1,"syntology":null},{"url":"/paper/deep-spiking-neural-networks-for-large","slug":"deep-spiking-neural-networks-for-large","title":"Deep Spiking Neural Networks for Large Vocabulary Automatic Speech Recognition","date":"2019-11-19","arxiv_id":"1911.08373","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-to-sequence-automatic-speech","slug":"sequence-to-sequence-automatic-speech","title":"Sequence-to-sequence Automatic Speech Recognition with Word Embedding Regularization and Fused Decoding","date":"2019-10-28","arxiv_id":"1910.12740","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-the-impact-of-speaker-localization","slug":"analyzing-the-impact-of-speaker-localization","title":"Analyzing the impact of speaker localization errors on speech separation for automatic speech recognition","date":"2019-10-24","arxiv_id":"1910.11114","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-end-to-end-speech-translation","slug":"multilingual-end-to-end-speech-translation","title":"Multilingual End-to-End Speech Translation","date":"2019-10-01","arxiv_id":"1910.00254","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multilingual-end-to-end-speech-translation#ran","syntology_url":"https://syntology.ai/paper/1910.00254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00254"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-rnn-transducer-modeling-for-end-to","slug":"improving-rnn-transducer-modeling-for-end-to","title":"Improving RNN Transducer Modeling for End-to-End Speech Recognition","date":"2019-09-26","arxiv_id":"1909.12415","repositories_listed":1,"syntology":null},{"url":"/paper/espresso-a-fast-end-to-end-neural-speech","slug":"espresso-a-fast-end-to-end-neural-speech","title":"Espresso: A Fast End-to-end Neural Speech Recognition Toolkit","date":"2019-09-18","arxiv_id":"1909.08723","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espresso-a-fast-end-to-end-neural-speech#ran","syntology_url":"https://syntology.ai/paper/1909.08723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.08723"}},"official":{"repos":["freewym/espresso"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/nemo-a-toolkit-for-building-ai-applications","slug":"nemo-a-toolkit-for-building-ai-applications","title":"NeMo: a toolkit for building AI applications using Neural Modules","date":"2019-09-14","arxiv_id":"1909.09577","repositories_listed":1,"syntology":null},{"url":"/paper/personalizing-asr-for-dysarthric-and-accented","slug":"personalizing-asr-for-dysarthric-and-accented","title":"Personalizing ASR for Dysarthric and Accented Speech with Limited Data","date":"2019-07-31","arxiv_id":"1907.13511","repositories_listed":1,"syntology":null},{"url":"/paper/mass-a-large-and-clean-multilingual-corpus-of","slug":"mass-a-large-and-clean-multilingual-corpus-of","title":"MaSS: A Large and Clean Multilingual Corpus of Sentence-aligned Spoken Utterances Extracted from the Bible","date":"2019-07-30","arxiv_id":"1907.12895","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-phonetic-and-graphemic","slug":"analyzing-phonetic-and-graphemic","title":"Analyzing Phonetic and Graphemic Representations in End-to-End Automatic Speech Recognition","date":"2019-07-09","arxiv_id":"1907.04224","repositories_listed":1,"syntology":null},{"url":"/paper/guided-source-separation-meets-a-strong-asr","slug":"guided-source-separation-meets-a-strong-asr","title":"Guided Source Separation Meets a Strong ASR Backend: Hitachi/Paderborn University Joint Investigation for Dinner Party ASR","date":"2019-05-29","arxiv_id":"1905.12230","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-for-audio-signal-processing","slug":"deep-learning-for-audio-signal-processing","title":"Deep Learning for Audio Signal Processing","date":"2019-04-30","arxiv_id":"1905.00078","repositories_listed":1,"syntology":null},{"url":"/paper/realizing-petabyte-scale-acoustic-modeling","slug":"realizing-petabyte-scale-acoustic-modeling","title":"Realizing Petabyte Scale Acoustic Modeling","date":"2019-04-24","arxiv_id":"1904.10584","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-speech-domain-adaptation-based","slug":"unsupervised-speech-domain-adaptation-based","title":"Unsupervised Speech Domain Adaptation Based on Disentangled Representation Learning for Robust Speech Recognition","date":"2019-04-12","arxiv_id":"1904.06086","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-language-intent-detection-using","slug":"spoken-language-intent-detection-using","title":"Spoken Language Intent Detection using Confusion2Vec","date":"2019-04-07","arxiv_id":"1904.03576","repositories_listed":1,"syntology":null},{"url":"/paper/imperceptible-robust-and-targeted-adversarial","slug":"imperceptible-robust-and-targeted-adversarial","title":"Imperceptible, Robust, and Targeted Adversarial Examples for Automatic Speech Recognition","date":"2019-03-22","arxiv_id":"1903.10346","repositories_listed":1,"syntology":null},{"url":"/paper/audiovisual-speaker-tracking-using-nonlinear","slug":"audiovisual-speaker-tracking-using-nonlinear","title":"Audiovisual Speaker Tracking using Nonlinear Dynamical Systems with Dynamic Stream Weights","date":"2019-03-14","arxiv_id":"1903.06031","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-effects-of-word","slug":"investigating-the-effects-of-word","title":"Investigating the Effects of Word Substitution Errors on Sentence Embeddings","date":"2018-11-16","arxiv_id":"1811.07021","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-grounding-for-sequence-to-sequence","slug":"multimodal-grounding-for-sequence-to-sequence","title":"Multimodal Grounding for Sequence-to-Sequence Speech Recognition","date":"2018-11-09","arxiv_id":"1811.03865","repositories_listed":1,"syntology":null},{"url":"/paper/language-modeling-for-code-switching","slug":"language-modeling-for-code-switching","title":"Language Modeling for Code-Switching: Evaluation, Integration of Monolingual Data, and Discriminative Training","date":"2018-10-28","arxiv_id":"1810.11895","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-on-high-resource-speech","slug":"pre-training-on-high-resource-speech","title":"Pre-training on high-resource speech recognition improves low-resource speech-to-text translation","date":"2018-09-05","arxiv_id":"1809.01431","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-adapt-a-meta-learning-approach","slug":"learning-to-adapt-a-meta-learning-approach","title":"Learning to adapt: a meta-learning approach for speaker adaptation","date":"2018-08-30","arxiv_id":"1808.10239","repositories_listed":1,"syntology":null},{"url":"/paper/wisebe-window-based-sentence-boundary","slug":"wisebe-window-based-sentence-boundary","title":"WiSeBE: Window-based Sentence Boundary Evaluation","date":"2018-08-27","arxiv_id":"1808.08850","repositories_listed":1,"syntology":null},{"url":"/paper/on-device-neural-language-model-based-word","slug":"on-device-neural-language-model-based-word","title":"On-Device Neural Language Model Based Word Prediction","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-techniques-for-language-model","slug":"a-comparison-of-techniques-for-language-model","title":"A Comparison of Techniques for Language Model Integration in Encoder-Decoder Speech Recognition","date":"2018-07-27","arxiv_id":"1807.10857","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-keyword-spotting-for-visual-speech","slug":"zero-shot-keyword-spotting-for-visual-speech","title":"Zero-shot keyword spotting for visual speech recognition in-the-wild","date":"2018-07-23","arxiv_id":"1807.08469","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-adaptation-techniques-and","slug":"a-comparison-of-adaptation-techniques-and","title":"A Comparison of Adaptation Techniques and Recurrent Neural Network Architectures","date":"2018-07-12","arxiv_id":"1807.06441","repositories_listed":1,"syntology":null},{"url":"/paper/word-error-rate-estimation-for-speech","slug":"word-error-rate-estimation-for-speech","title":"Word Error Rate Estimation for Speech Recognition: e-WER","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/quaternion-convolutional-neural-networks-for-1","slug":"quaternion-convolutional-neural-networks-for-1","title":"Quaternion Convolutional Neural Networks for End-to-End Automatic Speech Recognition","date":"2018-06-20","arxiv_id":"1806.07789","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-dnns-and-its-ensembles-on-the-timit","slug":"recurrent-dnns-and-its-ensembles-on-the-timit","title":"Recurrent DNNs and its Ensembles on the TIMIT Phone Recognition Task","date":"2018-06-19","arxiv_id":"1806.07186","repositories_listed":1,"syntology":null}],"record_sha256":"f6bc80d13f0a60e4933112a093c1d2b75385b07d0d6dc1a5213be38c048befa5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}