{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/11","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":58,"rows_per_page":100,"rows":[1001,1100],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/10","next":"/task/speech-recognition-1/papers/12","papers":[{"url":"/paper/joint-masked-cpc-and-ctc-training-for-asr","slug":"joint-masked-cpc-and-ctc-training-for-asr","title":"Joint Masked CPC and CTC Training for ASR","date":"2020-10-30","arxiv_id":"2011.00093","repositories_listed":1,"syntology":null},{"url":"/paper/speech-simclr-combining-contrastive-and","slug":"speech-simclr-combining-contrastive-and","title":"Speech SIMCLR: Combining Contrastive and Reconstruction Objective for Self-supervised Speech Representation Learning","date":"2020-10-27","arxiv_id":"2010.13991","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speech-simclr-combining-contrastive-and#ran","syntology_url":"https://syntology.ai/paper/2010.13991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13991"}},"official":{"repos":["athena-team/athena"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-end-to-end-multilingual-speech","slug":"large-scale-end-to-end-multilingual-speech","title":"Large-Scale End-to-End Multilingual Speech Recognition and Language Identification with Multi-Task Learning","date":"2020-10-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/probing-acoustic-representations-for-phonetic","slug":"probing-acoustic-representations-for-phonetic","title":"Probing Acoustic Representations for Phonetic Properties","date":"2020-10-25","arxiv_id":"2010.13007","repositories_listed":1,"syntology":null},{"url":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-end-to-end-speech","slug":"transformer-based-end-to-end-speech","title":"Transformer-based End-to-End Speech Recognition with Local Dense Synthesizer Attention","date":"2020-10-23","arxiv_id":"2010.12155","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-estimation-for-attention-based","slug":"confidence-estimation-for-attention-based","title":"Confidence Estimation for Attention-based Sequence-to-sequence Models for Speech Recognition","date":"2020-10-22","arxiv_id":"2010.11428","repositories_listed":1,"syntology":null},{"url":"/paper/how-phonotactics-affect-multilingual-and-zero","slug":"how-phonotactics-affect-multilingual-and-zero","title":"How Phonotactics Affect Multilingual and Zero-shot ASR Performance","date":"2020-10-22","arxiv_id":"2010.12104","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-evaluation-in-asr-are-our-models","slug":"rethinking-evaluation-in-asr-are-our-models","title":"Rethinking Evaluation in ASR: Are Our Models Robust Enough?","date":"2020-10-22","arxiv_id":"2010.11745","repositories_listed":1,"syntology":null},{"url":"/paper/emformer-efficient-memory-transformer-based","slug":"emformer-efficient-memory-transformer-based","title":"Emformer: Efficient Memory Transformer Based Acoustic Model For Low Latency Streaming Speech Recognition","date":"2020-10-21","arxiv_id":"2010.10759","repositories_listed":1,"syntology":null},{"url":"/paper/fastemit-low-latency-streaming-asr-with","slug":"fastemit-low-latency-streaming-asr-with","title":"FastEmit: Low-latency Streaming ASR with Sequence-level Emission Regularization","date":"2020-10-21","arxiv_id":"2010.11148","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/fastemit-low-latency-streaming-asr-with#ran","syntology_url":"https://syntology.ai/paper/2010.11148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11148"}},"official":null}},{"url":"/paper/towards-end-to-end-training-of-automatic","slug":"towards-end-to-end-training-of-automatic","title":"Towards End-to-End Training of Automatic Speech Recognition for Nigerian Pidgin","date":"2020-10-21","arxiv_id":"2010.11123","repositories_listed":1,"syntology":null},{"url":"/paper/venomave-clean-label-poisoning-against-speech","slug":"venomave-clean-label-poisoning-against-speech","title":"VenoMave: Targeted Poisoning Against Speech Recognition","date":"2020-10-21","arxiv_id":"2010.10682","repositories_listed":1,"syntology":null},{"url":"/paper/pushing-the-limits-of-semi-supervised","slug":"pushing-the-limits-of-semi-supervised","title":"Pushing the Limits of Semi-Supervised Learning for Automatic Speech Recognition","date":"2020-10-20","arxiv_id":"2010.10504","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pushing-the-limits-of-semi-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.10504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10504"}},"official":null}},{"url":"/paper/google-crowdsourced-speech-corpora-and","slug":"google-crowdsourced-speech-corpora-and","title":"Google Crowdsourced Speech Corpora and Related Open-Source Resources for Low-Resource Languages and Dialects: An Overview","date":"2020-10-14","arxiv_id":"2010.06778","repositories_listed":1,"syntology":null},{"url":"/paper/swiss-parliaments-corpus-an-automatically","slug":"swiss-parliaments-corpus-an-automatically","title":"Swiss Parliaments Corpus, an Automatically Aligned Swiss German Speech to Standard German Text Corpus","date":"2020-10-06","arxiv_id":"2010.02810","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-grounding-for-multimodal-speech","slug":"fine-grained-grounding-for-multimodal-speech","title":"Fine-Grained Grounding for Multimodal Speech Recognition","date":"2020-10-05","arxiv_id":"2010.02384","repositories_listed":1,"syntology":null},{"url":"/paper/online-neural-networks-for-change-point","slug":"online-neural-networks-for-change-point","title":"Online Neural Networks for Change-Point Detection","date":"2020-10-03","arxiv_id":"2010.01388","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-weighted-finite-state","slug":"differentiable-weighted-finite-state","title":"Differentiable Weighted Finite-State Transducers","date":"2020-10-02","arxiv_id":"2010.01003","repositories_listed":1,"syntology":null},{"url":"/paper/improving-vietnamese-named-entity-recognition","slug":"improving-vietnamese-named-entity-recognition","title":"Improving Vietnamese Named Entity Recognition from Speech Using Word Capitalization and Punctuation Recovery Models","date":"2020-10-01","arxiv_id":"2010.00198","repositories_listed":1,"syntology":null},{"url":"/paper/a-crowdsourced-open-source-kazakh-speech","slug":"a-crowdsourced-open-source-kazakh-speech","title":"A Crowdsourced Open-Source Kazakh Speech Corpus and Initial Speech Recognition Baseline","date":"2020-09-22","arxiv_id":"2009.10334","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-speech-2d-feature","slug":"end-to-end-learning-of-speech-2d-feature","title":"End-to-End Learning of Speech 2D Feature-Trajectory for Prosthetic Hands","date":"2020-09-22","arxiv_id":"2009.10283","repositories_listed":1,"syntology":null},{"url":"/paper/sdst-successive-decoding-for-speech-to-text","slug":"sdst-successive-decoding-for-speech-to-text","title":"Consecutive Decoding for Speech-to-text Translation","date":"2020-09-21","arxiv_id":"2009.09737","repositories_listed":1,"syntology":null},{"url":"/paper/any-to-many-voice-conversion-with-location","slug":"any-to-many-voice-conversion-with-location","title":"Any-to-Many Voice Conversion with Location-Relative Sequence-to-Sequence Modeling","date":"2020-09-06","arxiv_id":"2009.02725","repositories_listed":1,"syntology":null},{"url":"/paper/libri-adapt-a-new-speech-dataset-for","slug":"libri-adapt-a-new-speech-dataset-for","title":"Libri-Adapt: A New Speech Dataset for Unsupervised Domain Adaptation","date":"2020-09-06","arxiv_id":"2009.02814","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-deep-active-learning","slug":"a-survey-of-deep-active-learning","title":"A Survey of Deep Active Learning","date":"2020-08-30","arxiv_id":"2009.00236","repositories_listed":1,"syntology":null},{"url":"/paper/data-augmentation-using-prosody-and-false","slug":"data-augmentation-using-prosody-and-false","title":"Data augmentation using prosody and false starts to recognize non-native children's speech","date":"2020-08-29","arxiv_id":"2008.12914","repositories_listed":1,"syntology":null},{"url":"/paper/compiling-onnx-neural-network-models-using","slug":"compiling-onnx-neural-network-models-using","title":"Compiling ONNX Neural Network Models Using MLIR","date":"2020-08-19","arxiv_id":"2008.08272","repositories_listed":1,"syntology":null},{"url":"/paper/are-neural-open-domain-dialog-systems-robust","slug":"are-neural-open-domain-dialog-systems-robust","title":"Are Neural Open-Domain Dialog Systems Robust to Speech Recognition Errors in the Dialog History? An Empirical Study","date":"2020-08-18","arxiv_id":"2008.07683","repositories_listed":1,"syntology":null},{"url":"/paper/adaptation-algorithms-for-speech-recognition","slug":"adaptation-algorithms-for-speech-recognition","title":"Adaptation Algorithms for Neural Network-Based Speech Recognition: An Overview","date":"2020-08-14","arxiv_id":"2008.06580","repositories_listed":1,"syntology":null},{"url":"/paper/sum-product-networks-for-robust-automatic","slug":"sum-product-networks-for-robust-automatic","title":"Sum-Product Networks for Robust Automatic Speaker Identification","date":"2020-08-13","arxiv_id":"1910.11969","repositories_listed":1,"syntology":null},{"url":"/paper/investigation-of-end-to-end-speaker","slug":"investigation-of-end-to-end-speaker","title":"Investigation of End-To-End Speaker-Attributed ASR for Continuous Multi-Talker Recordings","date":"2020-08-11","arxiv_id":"2008.04546","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-the-knowledge-of-bert-for-sequence","slug":"distilling-the-knowledge-of-bert-for-sequence","title":"Distilling the Knowledge of BERT for Sequence-to-Sequence ASR","date":"2020-08-09","arxiv_id":"2008.03822","repositories_listed":1,"syntology":null},{"url":"/paper/word-error-rate-estimation-without-asr-output","slug":"word-error-rate-estimation-without-asr-output","title":"Word Error Rate Estimation Without ASR Output: e-WER2","date":"2020-08-08","arxiv_id":"2008.03403","repositories_listed":1,"syntology":null},{"url":"/paper/train-like-a-var-pro-efficient-training-of","slug":"train-like-a-var-pro-efficient-training-of","title":"Train Like a (Var)Pro: Efficient Training of Neural Networks with Variable Projection","date":"2020-07-26","arxiv_id":"2007.13171","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-transcription-and-translation-of","slug":"consistent-transcription-and-translation-of","title":"Consistent Transcription and Translation of Speech","date":"2020-07-24","arxiv_id":"2007.12741","repositories_listed":1,"syntology":null},{"url":"/paper/online-spatio-temporal-learning-in-deep","slug":"online-spatio-temporal-learning-in-deep","title":"Online Spatio-Temporal Learning in Deep Neural Networks","date":"2020-07-24","arxiv_id":"2007.12723","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-routing-framework-fully-capsule","slug":"sequential-routing-framework-fully-capsule","title":"Sequential Routing Framework: Fully Capsule Network-based Speech Recognition","date":"2020-07-23","arxiv_id":"2007.11747","repositories_listed":1,"syntology":null},{"url":"/paper/darf-a-data-reduced-fade-version-for","slug":"darf-a-data-reduced-fade-version-for","title":"DARF: A data-reduced FADE version for simulations of speech recognition thresholds with real hearing aids","date":"2020-07-10","arxiv_id":"2007.05378","repositories_listed":1,"syntology":null},{"url":"/paper/adascale-sgd-a-user-friendly-algorithm-for","slug":"adascale-sgd-a-user-friendly-algorithm-for","title":"AdaScale SGD: A User-Friendly Algorithm for Distributed Training","date":"2020-07-09","arxiv_id":"2007.05105","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/adascale-sgd-a-user-friendly-algorithm-for#ran","syntology_url":"https://syntology.ai/paper/2007.05105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.05105"}},"official":null}},{"url":"/paper/fast-transformers-with-clustered-attention","slug":"fast-transformers-with-clustered-attention","title":"Fast Transformers with Clustered Attention","date":"2020-07-09","arxiv_id":"2007.04825","repositories_listed":1,"syntology":null},{"url":"/paper/whole-word-segmental-speech-recognition-with","slug":"whole-word-segmental-speech-recognition-with","title":"Whole-Word Segmental Speech Recognition with Acoustic Word Embeddings","date":"2020-07-01","arxiv_id":"2007.00183","repositories_listed":1,"syntology":null},{"url":"/paper/incremental-training-of-a-recurrent-neural","slug":"incremental-training-of-a-recurrent-neural","title":"Incremental Training of a Recurrent Neural Network Exploiting a Multi-Scale Dynamic Memory","date":"2020-06-29","arxiv_id":"2006.16800","repositories_listed":1,"syntology":null},{"url":"/paper/avlnet-learning-audio-visual-language","slug":"avlnet-learning-audio-visual-language","title":"AVLnet: Learning Audio-Visual Language Representations from Instructional Videos","date":"2020-06-16","arxiv_id":"2006.09199","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-neural-architectures-trained","slug":"evaluation-of-neural-architectures-trained","title":"Evaluation of Neural Architectures Trained with Square Loss vs Cross-Entropy in Classification Tasks","date":"2020-06-12","arxiv_id":"2006.07322","repositories_listed":1,"syntology":null},{"url":"/paper/blissful-ignorance-anti-transfer-learning-for","slug":"blissful-ignorance-anti-transfer-learning-for","title":"Anti-Transfer Learning for Task Invariance in Convolutional Neural Networks for Speech Processing","date":"2020-06-11","arxiv_id":"2006.06494","repositories_listed":1,"syntology":null},{"url":"/paper/audino-a-modern-annotation-tool-for-audio-and","slug":"audino-a-modern-annotation-tool-for-audio-and","title":"audino: A Modern Annotation Tool for Audio and Speech","date":"2020-06-09","arxiv_id":"2006.05236","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-count-words-in-fluent-speech","slug":"learning-to-count-words-in-fluent-speech","title":"Learning to Count Words in Fluent Speech enables Online Speech Recognition","date":"2020-06-08","arxiv_id":"2006.04928","repositories_listed":1,"syntology":null},{"url":"/paper/improved-acoustic-word-embeddings-for-zero","slug":"improved-acoustic-word-embeddings-for-zero","title":"Improved acoustic word embeddings for zero-resource languages using multilingual transfer","date":"2020-06-02","arxiv_id":"2006.02295","repositories_listed":1,"syntology":null},{"url":"/paper/polydl-polyhedral-optimizations-for-creation","slug":"polydl-polyhedral-optimizations-for-creation","title":"PolyDL: Polyhedral Optimizations for Creation of High Performance DL primitives","date":"2020-06-02","arxiv_id":"2006.02230","repositories_listed":1,"syntology":null},{"url":"/paper/surprisal-triggered-conditional-computation","slug":"surprisal-triggered-conditional-computation","title":"Surprisal-Triggered Conditional Computation with Neural Networks","date":"2020-06-02","arxiv_id":"2006.01659","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-comparison-of-popular-end-to-end","slug":"on-the-comparison-of-popular-end-to-end","title":"On the Comparison of Popular End-to-End Models for Large Scale Speech Recognition","date":"2020-05-28","arxiv_id":"2005.14327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-comparison-of-popular-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2005.14327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14327"}},"official":{"repos":["cywang97/StreamingTransformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/subword-rnnlm-approximations-for-out-of","slug":"subword-rnnlm-approximations-for-out-of","title":"Subword RNNLM Approximations for Out-Of-Vocabulary Keyword Search","date":"2020-05-28","arxiv_id":"2005.13827","repositories_listed":1,"syntology":null},{"url":"/paper/cat-a-ctc-crf-based-asr-toolkit-bridging-the","slug":"cat-a-ctc-crf-based-asr-toolkit-bridging-the","title":"CAT: A CTC-CRF based ASR Toolkit Bridging the Hybrid and the End-to-end Approaches towards Data Efficiency and Low Latency","date":"2020-05-27","arxiv_id":"2005.13326","repositories_listed":1,"syntology":null},{"url":"/paper/phone-features-improve-speech-translation","slug":"phone-features-improve-speech-translation","title":"Phone Features Improve Speech Translation","date":"2020-05-27","arxiv_id":"2005.13681","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-end-to-end-speech-recognition-for","slug":"adapting-end-to-end-speech-recognition-for","title":"Adapting End-to-End Speech Recognition for Readable Subtitles","date":"2020-05-25","arxiv_id":"2005.12143","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-adversarial-examples-for-speech","slug":"detecting-adversarial-examples-for-speech","title":"Detecting Adversarial Examples for Speech Recognition via Uncertainty Quantification","date":"2020-05-24","arxiv_id":"2005.14611","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-named-entity-recognition-from","slug":"end-to-end-named-entity-recognition-from","title":"End-to-end Named Entity Recognition from English Speech","date":"2020-05-22","arxiv_id":"2005.11184","repositories_listed":1,"syntology":null},{"url":"/paper/low-latency-sequence-to-sequence-speech","slug":"low-latency-sequence-to-sequence-speech","title":"Low-Latency Sequence-to-Sequence Speech Recognition and Translation by Partial Hypothesis Selection","date":"2020-05-22","arxiv_id":"2005.11185","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/low-latency-sequence-to-sequence-speech#ran","syntology_url":"https://syntology.ai/paper/2005.11185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.11185"}},"official":{"repos":["dannigt/NMTGMinor.lowLatency"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-further-study-of-unsupervised-pre-training","slug":"a-further-study-of-unsupervised-pre-training","title":"A Further Study of Unsupervised Pre-training for Transformer Based Speech Recognition","date":"2020-05-20","arxiv_id":"2005.09862","repositories_listed":1,"syntology":null},{"url":"/paper/pychain-a-fully-parallelized-pytorch","slug":"pychain-a-fully-parallelized-pytorch","title":"PyChain: A Fully Parallelized PyTorch Implementation of LF-MMI for End-to-End ASR","date":"2020-05-20","arxiv_id":"2005.09824","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-monotonic-multihead-attention-for","slug":"enhancing-monotonic-multihead-attention-for","title":"Enhancing Monotonic Multihead Attention for Streaming ASR","date":"2020-05-19","arxiv_id":"2005.09394","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/enhancing-monotonic-multihead-attention-for#ran","syntology_url":"https://syntology.ai/paper/2005.09394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.09394"}},"official":{"repos":["hirofumi0810/neural_sp"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-adversarial-training-data","slug":"generative-adversarial-training-data","title":"Generative Adversarial Training Data Adaptation for Very Low-resource Automatic Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09256","repositories_listed":1,"syntology":null},{"url":"/paper/gev-beamforming-supported-by-doa-based-masks","slug":"gev-beamforming-supported-by-doa-based-masks","title":"GEV Beamforming Supported by DOA-based Masks Generated on Pairs of Microphones","date":"2020-05-19","arxiv_id":"2005.09587","repositories_listed":1,"syntology":null},{"url":"/paper/improved-noisy-student-training-for-automatic","slug":"improved-noisy-student-training-for-automatic","title":"Improved Noisy Student Training for Automatic Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09629","repositories_listed":1,"syntology":null},{"url":"/paper/investigations-on-phoneme-based-end-to-end","slug":"investigations-on-phoneme-based-end-to-end","title":"A systematic comparison of grapheme-based vs. phoneme-based label units for encoder-decoder-attention models","date":"2020-05-19","arxiv_id":"2005.09336","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-pseudo-labeling-for-speech","slug":"iterative-pseudo-labeling-for-speech","title":"Iterative Pseudo-Labeling for Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09267","repositories_listed":1,"syntology":null},{"url":"/paper/should-we-hard-code-the-recurrence-concept-or","slug":"should-we-hard-code-the-recurrence-concept-or","title":"Should we hard-code the recurrence concept or learn it instead ? Exploring the Transformer architecture for Audio-Visual Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09297","repositories_listed":1,"syntology":null},{"url":"/paper/quaternion-neural-networks-for-multi-channel","slug":"quaternion-neural-networks-for-multi-channel","title":"Quaternion Neural Networks for Multi-channel Distant Speech Recognition","date":"2020-05-18","arxiv_id":"2005.08566","repositories_listed":1,"syntology":null},{"url":"/paper/coupled-training-of-sequence-to-sequence","slug":"coupled-training-of-sequence-to-sequence","title":"Coupled Training of Sequence-to-Sequence Models for Accented Speech Recognition","date":"2020-05-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploring-tts-without-t-using-biologically","slug":"exploring-tts-without-t-using-biologically","title":"Exploring TTS without T Using Biologically/Psychologically Motivated Neural Network Modules (ZeroSpeech 2020)","date":"2020-05-11","arxiv_id":"2005.05487","repositories_listed":1,"syntology":null},{"url":"/paper/ctc-synchronous-training-for-monotonic","slug":"ctc-synchronous-training-for-monotonic","title":"CTC-synchronous Training for Monotonic Attention Model","date":"2020-05-10","arxiv_id":"2005.04712","repositories_listed":1,"syntology":null},{"url":"/paper/a-convolutional-neural-network-model-of-human","slug":"a-convolutional-neural-network-model-of-human","title":"A convolutional neural-network model of human cochlear mechanics and filter tuning for real-time applications","date":"2020-04-30","arxiv_id":"2004.14832","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-instructional-videos-probing-for-more","slug":"beyond-instructional-videos-probing-for-more","title":"Beyond Instructional Videos: Probing for More Diverse Visual-Textual Grounding on YouTube","date":"2020-04-29","arxiv_id":"2004.14338","repositories_listed":1,"syntology":null},{"url":"/paper/meta-transfer-learning-for-code-switched","slug":"meta-transfer-learning-for-code-switched","title":"Meta-Transfer Learning for Code-Switched Speech Recognition","date":"2020-04-29","arxiv_id":"2004.14228","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-competitive-end-to-end-speech","slug":"towards-a-competitive-end-to-end-speech","title":"Towards a Competitive End-to-End Speech Recognition for CHiME-6 Dinner Party Transcription","date":"2020-04-22","arxiv_id":"2004.10799","repositories_listed":1,"syntology":null},{"url":"/paper/espnet-st-all-in-one-speech-translation","slug":"espnet-st-all-in-one-speech-translation","title":"ESPnet-ST: All-in-One Speech Translation Toolkit","date":"2020-04-21","arxiv_id":"2004.10234","repositories_listed":1,"syntology":null},{"url":"/paper/clovacall-korean-goal-oriented-dialog-speech","slug":"clovacall-korean-goal-oriented-dialog-speech","title":"ClovaCall: Korean Goal-Oriented Dialog Speech Corpus for Automatic Speech Recognition of Contact Centers","date":"2020-04-20","arxiv_id":"2004.09367","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-teach-dnns-to-pay-attention-to-the","slug":"how-to-teach-dnns-to-pay-attention-to-the","title":"How to Teach DNNs to Pay Attention to the Visual Modality in Speech Recognition","date":"2020-04-17","arxiv_id":"2004.08250","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-grapheme-to-phoneme-1","slug":"transformer-based-grapheme-to-phoneme-1","title":"Transformer based Grapheme-to-Phoneme Conversion","date":"2020-04-14","arxiv_id":"2004.06338","repositories_listed":1,"syntology":null},{"url":"/paper/language-technology-programme-for-icelandic","slug":"language-technology-programme-for-icelandic","title":"Language Technology Programme for Icelandic 2019-2023","date":"2020-03-20","arxiv_id":"2003.09244","repositories_listed":1,"syntology":null},{"url":"/paper/can-we-read-speech-beyond-the-lips-rethinking","slug":"can-we-read-speech-beyond-the-lips-rethinking","title":"Can We Read Speech Beyond the Lips? Rethinking RoI Selection for Deep Visual Speech Recognition","date":"2020-03-06","arxiv_id":"2003.03206","repositories_listed":1,"syntology":null},{"url":"/paper/morfessor-emprune-improved-subword","slug":"morfessor-emprune-improved-subword","title":"Morfessor EM+Prune: Improved Subword Segmentation with Expectation Maximization and Pruning","date":"2020-03-06","arxiv_id":"2003.03131","repositories_listed":1,"syntology":null},{"url":"/paper/untangling-in-invariant-speech-recognition-1","slug":"untangling-in-invariant-speech-recognition-1","title":"Untangling in Invariant Speech Recognition","date":"2020-03-03","arxiv_id":"2003.01787","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-processing-advancements-by","slug":"natural-language-processing-advancements-by","title":"Natural Language Processing Advancements By Deep Learning: A Survey","date":"2020-03-02","arxiv_id":"2003.01200","repositories_listed":1,"syntology":null},{"url":"/paper/skinaugment-auto-encoding-speaker-conversions","slug":"skinaugment-auto-encoding-speaker-conversions","title":"SkinAugment: Auto-Encoding Speaker Conversions for Automatic Speech Translation","date":"2020-02-27","arxiv_id":"2002.12231","repositories_listed":1,"syntology":null},{"url":"/paper/universal-phone-recognition-with-a","slug":"universal-phone-recognition-with-a","title":"Universal Phone Recognition with a Multilingual Allophone System","date":"2020-02-26","arxiv_id":"2002.11800","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-speech-recognition-via-local","slug":"semi-supervised-speech-recognition-via-local","title":"Semi-Supervised Speech Recognition via Local Prior Matching","date":"2020-02-24","arxiv_id":"2002.10336","repositories_listed":1,"syntology":null},{"url":"/paper/imputer-sequence-modelling-via-imputation-and","slug":"imputer-sequence-modelling-via-imputation-and","title":"Imputer: Sequence Modelling via Imputation and Dynamic Programming","date":"2020-02-20","arxiv_id":"2002.08926","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imputer-sequence-modelling-via-imputation-and#ran","syntology_url":"https://syntology.ai/paper/2002.08926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08926"}},"official":null}},{"url":"/paper/uncertainty-in-structured-prediction","slug":"uncertainty-in-structured-prediction","title":"Uncertainty Estimation in Autoregressive Structured Prediction","date":"2020-02-18","arxiv_id":"2002.07650","repositories_listed":1,"syntology":null},{"url":"/paper/attentional-speech-recognition-models","slug":"attentional-speech-recognition-models","title":"Attentional Speech Recognition Models Misbehave on Out-of-domain Utterances","date":"2020-02-12","arxiv_id":"2002.05150","repositories_listed":1,"syntology":null},{"url":"/paper/vocoder-free-end-to-end-voice-conversion-with","slug":"vocoder-free-end-to-end-voice-conversion-with","title":"Vocoder-free End-to-End Voice Conversion with Transformer Network","date":"2020-02-05","arxiv_id":"2002.03808","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-speech-separation-dataset-and","slug":"continuous-speech-separation-dataset-and","title":"Continuous speech separation: dataset and analysis","date":"2020-01-30","arxiv_id":"2001.11482","repositories_listed":1,"syntology":null},{"url":"/paper/deep-xi-as-a-front-end-for-robust-automatic","slug":"deep-xi-as-a-front-end-for-robust-automatic","title":"Deep Xi as a Front-End for Robust Automatic Speech Recognition","date":"2020-01-28","arxiv_id":"1906.07319","repositories_listed":1,"syntology":null},{"url":"/paper/submodular-rank-aggregation-on-score-based","slug":"submodular-rank-aggregation-on-score-based","title":"Submodular Rank Aggregation on Score-based Permutations for Distributed Automatic Speech Recognition","date":"2020-01-27","arxiv_id":"2001.10529","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-self-supervised-learning-for-1","slug":"multi-task-self-supervised-learning-for-1","title":"Multi-task self-supervised learning for Robust Speech Recognition","date":"2020-01-25","arxiv_id":"2001.09239","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-labeling-approach-to-the-task-of","slug":"sequence-labeling-approach-to-the-task-of","title":"Sequence Labeling Approach to the Task of Sentence Boundary Detection","date":"2020-01-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-evolution-of-multi-turn","slug":"harnessing-evolution-of-multi-turn","title":"Harnessing Evolution of Multi-Turn Conversations for Effective Answer Retrieval","date":"2019-12-22","arxiv_id":"1912.10554","repositories_listed":1,"syntology":null},{"url":"/paper/generating-synthetic-audio-data-for-attention","slug":"generating-synthetic-audio-data-for-attention","title":"Generating Synthetic Audio Data for Attention-Based Speech Recognition Systems","date":"2019-12-19","arxiv_id":"1912.09257","repositories_listed":1,"syntology":null},{"url":"/paper/application-of-word2vec-in-phoneme","slug":"application-of-word2vec-in-phoneme","title":"Application of Word2vec in Phoneme Recognition","date":"2019-12-17","arxiv_id":"1912.08011","repositories_listed":1,"syntology":null}],"record_sha256":"6375b87e2ea9b4c54558783307449d1239bd21c72d723428864e5f8716ddc6ec","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}