{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/6","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":32,"rows_per_page":100,"rows":[501,600],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/5","next":"/task/automatic-speech-recognition-2/papers/7","papers":[{"url":"/paper/vietnamese-end-to-end-speech-recognition","slug":"vietnamese-end-to-end-speech-recognition","title":"Vietnamese end-to-end speech recognition using wav2vec 2.0","date":"2021-09-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficient-conformer-progressive-downsampling","slug":"efficient-conformer-progressive-downsampling","title":"Efficient conformer: Progressive downsampling and grouped attention for automatic speech recognition","date":"2021-08-31","arxiv_id":"2109.01163","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-with-joint","slug":"end-to-end-speech-recognition-with-joint","title":"End-to-End Speech Recognition With Joint Dereverberation Of Sub-Band Autoregressive Envelopes","date":"2021-08-09","arxiv_id":"2108.03975","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-distillation-from-bert-transformer","slug":"knowledge-distillation-from-bert-transformer","title":"Knowledge Distillation from BERT Transformer to Speech Transformer for Intent Classification","date":"2021-08-05","arxiv_id":"2108.02598","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-multilingual-end-to-end-speech","slug":"a-study-of-multilingual-end-to-end-speech","title":"A Study of Multilingual End-to-End Speech Recognition for Kazakh, Russian, and English","date":"2021-08-03","arxiv_id":"2108.01280","repositories_listed":1,"syntology":null},{"url":"/paper/the-history-of-speech-recognition-to-the-year","slug":"the-history-of-speech-recognition-to-the-year","title":"The History of Speech Recognition to the Year 2030","date":"2021-07-30","arxiv_id":"2108.00084","repositories_listed":1,"syntology":null},{"url":"/paper/usc-an-open-source-uzbek-speech-corpus-and","slug":"usc-an-open-source-uzbek-speech-corpus-and","title":"USC: An Open-Source Uzbek Speech Corpus and Initial Speech Recognition Experiments","date":"2021-07-30","arxiv_id":"2107.14419","repositories_listed":1,"syntology":null},{"url":"/paper/brazilian-portuguese-speech-recognition-using","slug":"brazilian-portuguese-speech-recognition-using","title":"Brazilian Portuguese Speech Recognition Using Wav2vec 2.0","date":"2021-07-23","arxiv_id":"2107.11414","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-model-with-self-adaptive-sliding","slug":"sequence-model-with-self-adaptive-sliding","title":"Sequence Model with Self-Adaptive Sliding Window for Efficient Spoken Document Segmentation","date":"2021-07-20","arxiv_id":"2107.09278","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-end-to-end-asr-based-on-blockwise","slug":"streaming-end-to-end-asr-based-on-blockwise","title":"Streaming End-to-End ASR based on Blockwise Non-Autoregressive Models","date":"2021-07-20","arxiv_id":"2107.09428","repositories_listed":1,"syntology":null},{"url":"/paper/token-level-supervised-contrastive-learning","slug":"token-level-supervised-contrastive-learning","title":"Token-Level Supervised Contrastive Learning for Punctuation Restoration","date":"2021-07-19","arxiv_id":"2107.09099","repositories_listed":1,"syntology":null},{"url":"/paper/strode-stochastic-boundary-ordinary","slug":"strode-stochastic-boundary-ordinary","title":"STRODE: Stochastic Boundary Ordinary Differential Equation","date":"2021-07-17","arxiv_id":"2107.08273","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/strode-stochastic-boundary-ordinary#ran","syntology_url":"https://syntology.ai/paper/2107.08273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08273"}},"official":{"repos":["Waffle-Liu/STRODE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/layer-wise-analysis-of-a-self-supervised","slug":"layer-wise-analysis-of-a-self-supervised","title":"Layer-wise Analysis of a Self-supervised Speech Representation Model","date":"2021-07-10","arxiv_id":"2107.04734","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-ctc-crf-based-end-to-end-speech","slug":"advancing-ctc-crf-based-end-to-end-speech","title":"Advancing CTC-CRF Based End-to-End Speech Recognition with Wordpieces and Conformers","date":"2021-07-07","arxiv_id":"2107.03007","repositories_listed":1,"syntology":null},{"url":"/paper/instant-one-shot-word-learning-for-context","slug":"instant-one-shot-word-learning-for-context","title":"Instant One-Shot Word-Learning for Context-Specific Neural Sequence-to-Sequence Speech Recognition","date":"2021-07-05","arxiv_id":"2107.02268","repositories_listed":1,"syntology":null},{"url":"/paper/tenet-a-time-reversal-enhancement-network-for","slug":"tenet-a-time-reversal-enhancement-network-for","title":"TENET: A Time-reversal Enhancement Network for Noise-robust ASR","date":"2021-07-04","arxiv_id":"2107.01531","repositories_listed":1,"syntology":null},{"url":"/paper/relaxed-attention-a-simple-method-to-boost","slug":"relaxed-attention-a-simple-method-to-boost","title":"Relaxed Attention: A Simple Method to Boost Performance of End-to-End Automatic Speech Recognition","date":"2021-07-02","arxiv_id":"2107.01275","repositories_listed":1,"syntology":null},{"url":"/paper/combining-frame-synchronous-and-label","slug":"combining-frame-synchronous-and-label","title":"Combining Frame-Synchronous and Label-Synchronous Systems for Speech Recognition","date":"2021-07-01","arxiv_id":"2107.00764","repositories_listed":1,"syntology":null},{"url":"/paper/pretext-tasks-selection-for-multitask-self","slug":"pretext-tasks-selection-for-multitask-self","title":"Pretext Tasks selection for multitask self-supervised speech representation learning","date":"2021-07-01","arxiv_id":"2107.00594","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pretext-tasks-selection-for-multitask-self#ran","syntology_url":"https://syntology.ai/paper/2107.00594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00594"}},"official":{"repos":["salah-zaiem/PL-groupselection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/momentum-pseudo-labeling-for-semi-supervised","slug":"momentum-pseudo-labeling-for-semi-supervised","title":"Momentum Pseudo-Labeling for Semi-Supervised Speech Recognition","date":"2021-06-16","arxiv_id":"2106.08922","repositories_listed":1,"syntology":null},{"url":"/paper/multi-speaker-asr-combining-non","slug":"multi-speaker-asr-combining-non","title":"Multi-Speaker ASR Combining Non-Autoregressive Conformer CTC and Conditional Speaker Chain","date":"2021-06-16","arxiv_id":"2106.08595","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-use-of-prosody-in-constituency","slug":"assessing-the-use-of-prosody-in-constituency","title":"Assessing the Use of Prosody in Constituency Parsing of Imperfect Transcripts","date":"2021-06-14","arxiv_id":"2106.07794","repositories_listed":1,"syntology":null},{"url":"/paper/learning-audio-visual-dereverberation","slug":"learning-audio-visual-dereverberation","title":"Learning Audio-Visual Dereverberation","date":"2021-06-14","arxiv_id":"2106.07732","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-external-pos-tagger-for","slug":"incorporating-external-pos-tagger-for","title":"Incorporating External POS Tagger for Punctuation Restoration","date":"2021-06-12","arxiv_id":"2106.06731","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-contextual-language-model","slug":"attention-based-contextual-language-model","title":"Attention-based Contextual Language Model Adaptation for Speech Recognition","date":"2021-06-02","arxiv_id":"2106.01451","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/attention-based-contextual-language-model#ran","syntology_url":"https://syntology.ai/paper/2106.01451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01451"}},"official":{"repos":["amazon-research/contextual-attention-nlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-speech-recognition-in-sanskrit-a","slug":"automatic-speech-recognition-in-sanskrit-a","title":"Automatic Speech Recognition in Sanskrit: A New Speech Corpus and Modelling Insights","date":"2021-06-02","arxiv_id":"2106.05852","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-reordering-capability-in","slug":"investigating-the-reordering-capability-in","title":"Investigating the Reordering Capability in CTC-based Non-Autoregressive End-to-End Speech Translation","date":"2021-05-11","arxiv_id":"2105.04840","repositories_listed":1,"syntology":null},{"url":"/paper/fastcorrect-fast-error-correction-with-edit","slug":"fastcorrect-fast-error-correction-with-edit","title":"FastCorrect: Fast Error Correction with Edit Alignment for Automatic Speech Recognition","date":"2021-05-09","arxiv_id":"2105.03842","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/fastcorrect-fast-error-correction-with-edit#ran","syntology_url":"https://syntology.ai/paper/2105.03842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03842"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/end-to-end-speech-recognition-from-federated","slug":"end-to-end-speech-recognition-from-federated","title":"End-to-End Speech Recognition from Federated Acoustic Models","date":"2021-04-29","arxiv_id":"2104.14297","repositories_listed":1,"syntology":null},{"url":"/paper/lebenchmark-a-reproducible-framework-for","slug":"lebenchmark-a-reproducible-framework-for","title":"LeBenchmark: A Reproducible Framework for Assessing Self-Supervised Representation Learning from Speech","date":"2021-04-23","arxiv_id":"2104.11462","repositories_listed":1,"syntology":null},{"url":"/paper/a-method-to-reveal-speaker-identity-in","slug":"a-method-to-reveal-speaker-identity-in","title":"A Method to Reveal Speaker Identity in Distributed ASR Training, and How to Counter It","date":"2021-04-15","arxiv_id":"2104.07815","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-independence-for-pretext-task","slug":"conditional-independence-for-pretext-task","title":"Conditional independence for pretext task selection in Self-supervised speech representation learning","date":"2021-04-15","arxiv_id":"2104.07388","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-speech-recognition-with","slug":"cross-domain-speech-recognition-with","title":"Cross-domain Speech Recognition with Unsupervised Character-level Distribution Matching","date":"2021-04-15","arxiv_id":"2104.07491","repositories_listed":1,"syntology":null},{"url":"/paper/nemo-inverse-text-normalization-from","slug":"nemo-inverse-text-normalization-from","title":"NeMo Inverse Text Normalization: From Development To Production","date":"2021-04-11","arxiv_id":"2104.05055","repositories_listed":1,"syntology":null},{"url":"/paper/nemo-toolbox-for-speech-dataset-construction","slug":"nemo-toolbox-for-speech-dataset-construction","title":"A Toolbox for Construction and Analysis of Speech Datasets","date":"2021-04-11","arxiv_id":"2104.04896","repositories_listed":1,"syntology":null},{"url":"/paper/rnn-transducer-models-for-spoken-language","slug":"rnn-transducer-models-for-spoken-language","title":"RNN Transducer Models For Spoken Language Understanding","date":"2021-04-08","arxiv_id":"2104.03842","repositories_listed":1,"syntology":null},{"url":"/paper/speak-or-chat-with-me-end-to-end-spoken","slug":"speak-or-chat-with-me-end-to-end-spoken","title":"Speak or Chat with Me: End-to-End Spoken Language Understanding System with Flexible Inputs","date":"2021-04-07","arxiv_id":"2104.05752","repositories_listed":1,"syntology":null},{"url":"/paper/lt-lm-a-novel-non-autoregressive-language","slug":"lt-lm-a-novel-non-autoregressive-language","title":"LT-LM: a novel non-autoregressive language model for single-shot lattice rescoring","date":"2021-04-06","arxiv_id":"2104.02526","repositories_listed":1,"syntology":null},{"url":"/paper/exkaldi-rt-a-real-time-automatic-speech","slug":"exkaldi-rt-a-real-time-automatic-speech","title":"ExKaldi-RT: A Real-Time Automatic Speech Recognition Extension Toolkit of Kaldi","date":"2021-04-03","arxiv_id":"2104.01384","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-fly-aligned-data-augmentation-for","slug":"on-the-fly-aligned-data-augmentation-for","title":"On-the-Fly Aligned Data Augmentation for Sequence-to-Sequence ASR","date":"2021-04-03","arxiv_id":"2104.01393","repositories_listed":1,"syntology":null},{"url":"/paper/q-asr-integer-only-zero-shot-quantization-for","slug":"q-asr-integer-only-zero-shot-quantization-for","title":"Integer-only Zero-shot Quantization for Efficient Speech Recognition","date":"2021-03-31","arxiv_id":"2103.16827","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-bias-in-automatic-speech","slug":"quantifying-bias-in-automatic-speech","title":"Quantifying Bias in Automatic Speech Recognition","date":"2021-03-28","arxiv_id":"2103.15122","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-neural-representations-for","slug":"leveraging-neural-representations-for","title":"Leveraging pre-trained representations to improve access to untranscribed speech from endangered languages","date":"2021-03-26","arxiv_id":"2103.14583","repositories_listed":1,"syntology":null},{"url":"/paper/sok-a-modularized-approach-to-study-the","slug":"sok-a-modularized-approach-to-study-the","title":"SoK: A Modularized Approach to Study the Security of Automatic Speech Recognition Systems","date":"2021-03-19","arxiv_id":"2103.10651","repositories_listed":1,"syntology":null},{"url":"/paper/fast-development-of-asr-in-african-languages","slug":"fast-development-of-asr-in-african-languages","title":"Fast Development of ASR in African Languages using Self Supervised Speech Representation Learning","date":"2021-03-16","arxiv_id":"2103.08993","repositories_listed":1,"syntology":null},{"url":"/paper/a-parallelizable-lattice-rescoring-strategy","slug":"a-parallelizable-lattice-rescoring-strategy","title":"A Parallelizable Lattice Rescoring Strategy with Neural Language Models","date":"2021-03-08","arxiv_id":"2103.05081","repositories_listed":1,"syntology":null},{"url":"/paper/waveguard-understanding-and-mitigating-audio","slug":"waveguard-understanding-and-mitigating-audio","title":"WaveGuard: Understanding and Mitigating Audio Adversarial Examples","date":"2021-03-04","arxiv_id":"2103.03344","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/waveguard-understanding-and-mitigating-audio#ran","syntology_url":"https://syntology.ai/paper/2103.03344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.03344"}},"official":null}},{"url":"/paper/exploiting-attention-based-sequence-to","slug":"exploiting-attention-based-sequence-to","title":"Exploiting Attention-based Sequence-to-Sequence Architectures for Sound Event Localization","date":"2021-02-28","arxiv_id":"2103.00417","repositories_listed":1,"syntology":null},{"url":"/paper/data-fusion-for-audiovisual-speaker","slug":"data-fusion-for-audiovisual-speaker","title":"Data Fusion for Audiovisual Speaker Localization: Extending Dynamic Stream Weights to the Spatial Domain","date":"2021-02-23","arxiv_id":"2102.11588","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-phonetic-neural-model-for-correction","slug":"hybrid-phonetic-neural-model-for-correction","title":"Hybrid phonetic-neural model for correction in speech recognition systems","date":"2021-02-12","arxiv_id":"2102.06744","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-language-models-with-lstm-based","slug":"transformer-language-models-with-lstm-based","title":"Transformer Language Models with LSTM-based Cross-utterance Information Representation","date":"2021-02-12","arxiv_id":"2102.06474","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-end-to-end-models-for","slug":"an-investigation-of-end-to-end-models-for","title":"An Investigation of End-to-End Models for Robust Speech Recognition","date":"2021-02-11","arxiv_id":"2102.06237","repositories_listed":1,"syntology":null},{"url":"/paper/dompteur-taming-audio-adversarial-examples","slug":"dompteur-taming-audio-adversarial-examples","title":"Dompteur: Taming Audio Adversarial Examples","date":"2021-02-10","arxiv_id":"2102.05431","repositories_listed":1,"syntology":null},{"url":"/paper/effects-of-layer-freezing-when-transferring","slug":"effects-of-layer-freezing-when-transferring","title":"Effects of Layer Freezing on Transferring a Speech Recognition System to Under-resourced Languages","date":"2021-02-08","arxiv_id":"2102.04097","repositories_listed":1,"syntology":null},{"url":"/paper/confusion2vec-2-0-enriching-ambiguous-spoken","slug":"confusion2vec-2-0-enriching-ambiguous-spoken","title":"Confusion2vec 2.0: Enriching Ambiguous Spoken Language Representations with Subwords","date":"2021-02-03","arxiv_id":"2102.02270","repositories_listed":1,"syntology":null},{"url":"/paper/bendr-using-transformers-and-a-contrastive","slug":"bendr-using-transformers-and-a-contrastive","title":"BENDR: using transformers and a contrastive self-supervised learning task to learn from massive amounts of EEG data","date":"2021-01-28","arxiv_id":"2101.12037","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-speech-recognition-by-end-to-end","slug":"arabic-speech-recognition-by-end-to-end","title":"Arabic Speech Recognition by End-to-End, Modular Systems and Human","date":"2021-01-21","arxiv_id":"2101.08454","repositories_listed":1,"syntology":null},{"url":"/paper/av-taris-online-audio-visual-speech","slug":"av-taris-online-audio-visual-speech","title":"AV Taris: Online Audio-Visual Speech Recognition","date":"2020-12-14","arxiv_id":"2012.07467","repositories_listed":1,"syntology":null},{"url":"/paper/on-knowledge-distillation-for-direct-speech","slug":"on-knowledge-distillation-for-direct-speech","title":"On Knowledge Distillation for Direct Speech Translation","date":"2020-12-09","arxiv_id":"2012.04964","repositories_listed":1,"syntology":null},{"url":"/paper/mls-a-large-scale-multilingual-dataset-for","slug":"mls-a-large-scale-multilingual-dataset-for","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","date":"2020-12-07","arxiv_id":"2012.03411","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-asr-system-with-automatic","slug":"end-to-end-asr-system-with-automatic","title":"End to End ASR System with Automatic Punctuation Insertion","date":"2020-12-03","arxiv_id":"2012.02012","repositories_listed":1,"syntology":null},{"url":"/paper/attentively-embracing-noise-for-robust-latent","slug":"attentively-embracing-noise-for-robust-latent","title":"Attentively Embracing Noise for Robust Latent Representation in BERT","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-automatic-speech-recognition-for","slug":"end-to-end-automatic-speech-recognition-for","title":"End-to-End Automatic Speech Recognition for Gujarati","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/metacat-a-metadata-based-task-oriented","slug":"metacat-a-metadata-based-task-oriented","title":"metaCAT: A Metadata-based Task-oriented Chatbot Annotation Tool","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/wpd-an-improved-neural-beamformer-for","slug":"wpd-an-improved-neural-beamformer-for","title":"WPD++: An Improved Neural Beamformer for Simultaneous Speech Separation and Dereverberation","date":"2020-11-18","arxiv_id":"2011.09162","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-neural-architecture-search-for-end","slug":"efficient-neural-architecture-search-for-end","title":"Efficient Neural Architecture Search for End-to-end Speech Recognition via Straight-Through Gradients","date":"2020-11-11","arxiv_id":"2011.05649","repositories_listed":1,"syntology":null},{"url":"/paper/improving-rnn-transducer-based-asr-with","slug":"improving-rnn-transducer-based-asr-with","title":"Improving RNN Transducer Based ASR with Auxiliary Tasks","date":"2020-11-05","arxiv_id":"2011.03109","repositories_listed":1,"syntology":null},{"url":"/paper/minimum-bayes-risk-training-for-end-to-end","slug":"minimum-bayes-risk-training-for-end-to-end","title":"Minimum Bayes Risk Training for End-to-End Speaker-Attributed ASR","date":"2020-11-03","arxiv_id":"2011.02921","repositories_listed":1,"syntology":null},{"url":"/paper/dual-decoder-transformer-for-joint-automatic","slug":"dual-decoder-transformer-for-joint-automatic","title":"Dual-decoder Transformer for Joint Automatic Speech Recognition and Multilingual Speech Translation","date":"2020-11-02","arxiv_id":"2011.00747","repositories_listed":1,"syntology":null},{"url":"/paper/direct-segmentation-models-for-streaming","slug":"direct-segmentation-models-for-streaming","title":"Direct Segmentation Models for Streaming Speech Translation","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/punctuation-restoration-using-transformer","slug":"punctuation-restoration-using-transformer","title":"Punctuation Restoration using Transformer Models for High-and Low-Resource Languages","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-bottleneck-features-for","slug":"multilingual-bottleneck-features-for","title":"Multilingual Bottleneck Features for Improving ASR Performance of Code-Switched Speech in Under-Resourced Languages","date":"2020-10-31","arxiv_id":"2011.03118","repositories_listed":1,"syntology":null},{"url":"/paper/joint-masked-cpc-and-ctc-training-for-asr","slug":"joint-masked-cpc-and-ctc-training-for-asr","title":"Joint Masked CPC and CTC Training for ASR","date":"2020-10-30","arxiv_id":"2011.00093","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-end-to-end-multilingual-speech","slug":"large-scale-end-to-end-multilingual-speech","title":"Large-Scale End-to-End Multilingual Speech Recognition and Language Identification with Multi-Task Learning","date":"2020-10-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-estimation-for-attention-based","slug":"confidence-estimation-for-attention-based","title":"Confidence Estimation for Attention-based Sequence-to-sequence Models for Speech Recognition","date":"2020-10-22","arxiv_id":"2010.11428","repositories_listed":1,"syntology":null},{"url":"/paper/how-phonotactics-affect-multilingual-and-zero","slug":"how-phonotactics-affect-multilingual-and-zero","title":"How Phonotactics Affect Multilingual and Zero-shot ASR Performance","date":"2020-10-22","arxiv_id":"2010.12104","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-evaluation-in-asr-are-our-models","slug":"rethinking-evaluation-in-asr-are-our-models","title":"Rethinking Evaluation in ASR: Are Our Models Robust Enough?","date":"2020-10-22","arxiv_id":"2010.11745","repositories_listed":1,"syntology":null},{"url":"/paper/fastemit-low-latency-streaming-asr-with","slug":"fastemit-low-latency-streaming-asr-with","title":"FastEmit: Low-latency Streaming ASR with Sequence-level Emission Regularization","date":"2020-10-21","arxiv_id":"2010.11148","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/fastemit-low-latency-streaming-asr-with#ran","syntology_url":"https://syntology.ai/paper/2010.11148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11148"}},"official":null}},{"url":"/paper/towards-end-to-end-training-of-automatic","slug":"towards-end-to-end-training-of-automatic","title":"Towards End-to-End Training of Automatic Speech Recognition for Nigerian Pidgin","date":"2020-10-21","arxiv_id":"2010.11123","repositories_listed":1,"syntology":null},{"url":"/paper/venomave-clean-label-poisoning-against-speech","slug":"venomave-clean-label-poisoning-against-speech","title":"VenoMave: Targeted Poisoning Against Speech Recognition","date":"2020-10-21","arxiv_id":"2010.10682","repositories_listed":1,"syntology":null},{"url":"/paper/pushing-the-limits-of-semi-supervised","slug":"pushing-the-limits-of-semi-supervised","title":"Pushing the Limits of Semi-Supervised Learning for Automatic Speech Recognition","date":"2020-10-20","arxiv_id":"2010.10504","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pushing-the-limits-of-semi-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.10504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10504"}},"official":null}},{"url":"/paper/google-crowdsourced-speech-corpora-and","slug":"google-crowdsourced-speech-corpora-and","title":"Google Crowdsourced Speech Corpora and Related Open-Source Resources for Low-Resource Languages and Dialects: An Overview","date":"2020-10-14","arxiv_id":"2010.06778","repositories_listed":1,"syntology":null},{"url":"/paper/swiss-parliaments-corpus-an-automatically","slug":"swiss-parliaments-corpus-an-automatically","title":"Swiss Parliaments Corpus, an Automatically Aligned Swiss German Speech to Standard German Text Corpus","date":"2020-10-06","arxiv_id":"2010.02810","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-grounding-for-multimodal-speech","slug":"fine-grained-grounding-for-multimodal-speech","title":"Fine-Grained Grounding for Multimodal Speech Recognition","date":"2020-10-05","arxiv_id":"2010.02384","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-speech-2d-feature","slug":"end-to-end-learning-of-speech-2d-feature","title":"End-to-End Learning of Speech 2D Feature-Trajectory for Prosthetic Hands","date":"2020-09-22","arxiv_id":"2009.10283","repositories_listed":1,"syntology":null},{"url":"/paper/data-augmentation-using-prosody-and-false","slug":"data-augmentation-using-prosody-and-false","title":"Data augmentation using prosody and false starts to recognize non-native children's speech","date":"2020-08-29","arxiv_id":"2008.12914","repositories_listed":1,"syntology":null},{"url":"/paper/are-neural-open-domain-dialog-systems-robust","slug":"are-neural-open-domain-dialog-systems-robust","title":"Are Neural Open-Domain Dialog Systems Robust to Speech Recognition Errors in the Dialog History? An Empirical Study","date":"2020-08-18","arxiv_id":"2008.07683","repositories_listed":1,"syntology":null},{"url":"/paper/sum-product-networks-for-robust-automatic","slug":"sum-product-networks-for-robust-automatic","title":"Sum-Product Networks for Robust Automatic Speaker Identification","date":"2020-08-13","arxiv_id":"1910.11969","repositories_listed":1,"syntology":null},{"url":"/paper/investigation-of-end-to-end-speaker","slug":"investigation-of-end-to-end-speaker","title":"Investigation of End-To-End Speaker-Attributed ASR for Continuous Multi-Talker Recordings","date":"2020-08-11","arxiv_id":"2008.04546","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-the-knowledge-of-bert-for-sequence","slug":"distilling-the-knowledge-of-bert-for-sequence","title":"Distilling the Knowledge of BERT for Sequence-to-Sequence ASR","date":"2020-08-09","arxiv_id":"2008.03822","repositories_listed":1,"syntology":null},{"url":"/paper/word-error-rate-estimation-without-asr-output","slug":"word-error-rate-estimation-without-asr-output","title":"Word Error Rate Estimation Without ASR Output: e-WER2","date":"2020-08-08","arxiv_id":"2008.03403","repositories_listed":1,"syntology":null},{"url":"/paper/fast-transformers-with-clustered-attention","slug":"fast-transformers-with-clustered-attention","title":"Fast Transformers with Clustered Attention","date":"2020-07-09","arxiv_id":"2007.04825","repositories_listed":1,"syntology":null},{"url":"/paper/avlnet-learning-audio-visual-language","slug":"avlnet-learning-audio-visual-language","title":"AVLnet: Learning Audio-Visual Language Representations from Instructional Videos","date":"2020-06-16","arxiv_id":"2006.09199","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-neural-architectures-trained","slug":"evaluation-of-neural-architectures-trained","title":"Evaluation of Neural Architectures Trained with Square Loss vs Cross-Entropy in Classification Tasks","date":"2020-06-12","arxiv_id":"2006.07322","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-count-words-in-fluent-speech","slug":"learning-to-count-words-in-fluent-speech","title":"Learning to Count Words in Fluent Speech enables Online Speech Recognition","date":"2020-06-08","arxiv_id":"2006.04928","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-comparison-of-popular-end-to-end","slug":"on-the-comparison-of-popular-end-to-end","title":"On the Comparison of Popular End-to-End Models for Large Scale Speech Recognition","date":"2020-05-28","arxiv_id":"2005.14327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-comparison-of-popular-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2005.14327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14327"}},"official":{"repos":["cywang97/StreamingTransformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adapting-end-to-end-speech-recognition-for","slug":"adapting-end-to-end-speech-recognition-for","title":"Adapting End-to-End Speech Recognition for Readable Subtitles","date":"2020-05-25","arxiv_id":"2005.12143","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-adversarial-examples-for-speech","slug":"detecting-adversarial-examples-for-speech","title":"Detecting Adversarial Examples for Speech Recognition via Uncertainty Quantification","date":"2020-05-24","arxiv_id":"2005.14611","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-named-entity-recognition-from","slug":"end-to-end-named-entity-recognition-from","title":"End-to-end Named Entity Recognition from English Speech","date":"2020-05-22","arxiv_id":"2005.11184","repositories_listed":1,"syntology":null}],"record_sha256":"06edc10c5a58b94a60de0fe69e894003b172b5ef5ee10a4602191deaa90f4b8b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}