{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/5","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":31,"rows_per_page":100,"rows":[401,500],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/4","next":"/task/automatic-speech-recognition/papers/6","papers":[{"url":"/paper/speech-enhanced-and-noise-aware-networks-for","slug":"speech-enhanced-and-noise-aware-networks-for","title":"Speech-enhanced and Noise-aware Networks for Robust Speech Recognition","date":"2022-03-25","arxiv_id":"2203.13696","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-for-speech","slug":"automatic-speech-recognition-for-speech","title":"Automatic Speech Recognition for Speech Assessment of Persian Preschool Children","date":"2022-03-24","arxiv_id":"2203.12886","repositories_listed":1,"syntology":null},{"url":"/paper/neural-predictor-for-black-box-adversarial","slug":"neural-predictor-for-black-box-adversarial","title":"Neural Predictor for Black-Box Adversarial Attacks on Speech Recognition","date":"2022-03-18","arxiv_id":"2203.09849","repositories_listed":1,"syntology":null},{"url":"/paper/red-ace-robust-error-detection-for-asr-using-1","slug":"red-ace-robust-error-detection-for-asr-using-1","title":"RED-ACE: Robust Error Detection for ASR using Confidence Embeddings","date":"2022-03-14","arxiv_id":"2203.07172","repositories_listed":1,"syntology":null},{"url":"/paper/dual-textless-spoken-question-answering-with-1","slug":"dual-textless-spoken-question-answering-with-1","title":"DUAL: Discrete Spoken Unit Adaptive Learning for Textless Spoken Question Answering","date":"2022-03-09","arxiv_id":"2203.04911","repositories_listed":1,"syntology":null},{"url":"/paper/towards-contextual-spelling-correction-for","slug":"towards-contextual-spelling-correction-for","title":"Towards Contextual Spelling Correction for Customization of End-to-end Speech Recognition Systems","date":"2022-03-02","arxiv_id":"2203.00888","repositories_listed":1,"syntology":null},{"url":"/paper/sentiment-word-aware-multimodal-refinement","slug":"sentiment-word-aware-multimodal-refinement","title":"Sentiment Word Aware Multimodal Refinement for Multimodal Sentiment Analysis with ASR Errors","date":"2022-03-01","arxiv_id":"2203.00257","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-uni-modal-self-supervised-learning-1","slug":"leveraging-uni-modal-self-supervised-learning-1","title":"Leveraging Unimodal Self-Supervised Learning for Multimodal Audio-Visual Speech Recognition","date":"2022-02-24","arxiv_id":"2203.07996","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/leveraging-uni-modal-self-supervised-learning-1#ran","syntology_url":"https://syntology.ai/paper/2203.07996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07996"}},"official":{"repos":["lumia-group/leveraging-self-supervised-learning-for-avsr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-ctc-based-speech-recognition-via","slug":"improving-ctc-based-speech-recognition-via","title":"Improving CTC-based speech recognition via knowledge transferring from pre-trained language models","date":"2022-02-22","arxiv_id":"2203.03582","repositories_listed":1,"syntology":null},{"url":"/paper/aishell-ner-named-entity-recognition-from","slug":"aishell-ner-named-entity-recognition-from","title":"AISHELL-NER: Named Entity Recognition from Chinese Speech","date":"2022-02-17","arxiv_id":"2202.08533","repositories_listed":1,"syntology":null},{"url":"/paper/adima-abuse-detection-in-multilingual-audio","slug":"adima-abuse-detection-in-multilingual-audio","title":"ADIMA: Abuse Detection In Multilingual Audio","date":"2022-02-16","arxiv_id":"2202.07991","repositories_listed":1,"syntology":null},{"url":"/paper/improving-automatic-speech-recognition-for","slug":"improving-automatic-speech-recognition-for","title":"Improving Automatic Speech Recognition for Non-Native English with Transfer Learning and Language Model Decoding","date":"2022-02-10","arxiv_id":"2202.05209","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-adapter-transfer-of-self-supervised","slug":"efficient-adapter-transfer-of-self-supervised","title":"Efficient Adapter Transfer of Self-Supervised Speech Models for Automatic Speech Recognition","date":"2022-02-07","arxiv_id":"2202.03218","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-adapter-transfer-of-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2202.03218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03218"}},"official":null}},{"url":"/paper/streaming-multi-talker-asr-with-token-level","slug":"streaming-multi-talker-asr-with-token-level","title":"Streaming Multi-Talker ASR with Token-Level Serialized Output Training","date":"2022-02-02","arxiv_id":"2202.00842","repositories_listed":1,"syntology":null},{"url":"/paper/star-temporal-classification-sequence","slug":"star-temporal-classification-sequence","title":"Star Temporal Classification: Sequence Classification with Partially Labeled Data","date":"2022-01-28","arxiv_id":"2201.12208","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-phonetic-inventories-with","slug":"discovering-phonetic-inventories-with","title":"Discovering Phonetic Inventories with Crosslingual Automatic Speech Recognition","date":"2022-01-26","arxiv_id":"2201.11207","repositories_listed":1,"syntology":null},{"url":"/paper/unified-multimodal-punctuation-restoration","slug":"unified-multimodal-punctuation-restoration","title":"Unified Multimodal Punctuation Restoration Framework for Mixed-Modality Corpus","date":"2022-01-24","arxiv_id":"2202.00468","repositories_listed":1,"syntology":null},{"url":"/paper/neural-architecture-search-for-lf-mmi-trained","slug":"neural-architecture-search-for-lf-mmi-trained","title":"Neural Architecture Search For LF-MMI Trained Time Delay Neural Networks","date":"2022-01-08","arxiv_id":"2201.03943","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-datasets-in","slug":"automatic-speech-recognition-datasets-in","title":"Automatic Speech Recognition Datasets in Cantonese: A Survey and New Dataset","date":"2022-01-07","arxiv_id":"2201.02419","repositories_listed":1,"syntology":null},{"url":"/paper/improving-mandarin-end-to-end-speech","slug":"improving-mandarin-end-to-end-speech","title":"Improving Mandarin End-to-End Speech Recognition with Word N-gram Language Model","date":"2022-01-06","arxiv_id":"2201.01995","repositories_listed":1,"syntology":null},{"url":"/paper/robust-self-supervised-audio-visual-speech","slug":"robust-self-supervised-audio-visual-speech","title":"Robust Self-Supervised Audio-Visual Speech Recognition","date":"2022-01-05","arxiv_id":"2201.01763","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-self-supervised-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2201.01763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01763"}},"official":{"repos":["facebookresearch/av_hubert"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/regularizing-end-to-end-speech-translation","slug":"regularizing-end-to-end-speech-translation","title":"Regularizing End-to-End Speech Translation with Triangular Decomposition Agreement","date":"2021-12-21","arxiv_id":"2112.10991","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-for-monolingual-end-to-end","slug":"continual-learning-for-monolingual-end-to-end","title":"Continual Learning for Monolingual End-to-End Automatic Speech Recognition","date":"2021-12-17","arxiv_id":"2112.09427","repositories_listed":1,"syntology":null},{"url":"/paper/x-vector-based-voice-activity-detection-for-1","slug":"x-vector-based-voice-activity-detection-for-1","title":"X-Vector based voice activity detection for multi-genre broadcast speech-to-text","date":"2021-12-09","arxiv_id":"2112.05016","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-training-and-decoding-for-end-to","slug":"consistent-training-and-decoding-for-end-to","title":"Consistent Training and Decoding For End-to-end Speech Recognition Using Lattice-free MMI","date":"2021-12-05","arxiv_id":"2112.02498","repositories_listed":1,"syntology":null},{"url":"/paper/slue-new-benchmark-tasks-for-spoken-language","slug":"slue-new-benchmark-tasks-for-spoken-language","title":"SLUE: New Benchmark Tasks for Spoken Language Understanding Evaluation on Natural Speech","date":"2021-11-19","arxiv_id":"2111.10367","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slue-new-benchmark-tasks-for-spoken-language#ran","syntology_url":"https://syntology.ai/paper/2111.10367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10367"}},"official":{"repos":["asappresearch/slue-toolkit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-randomized-smoothing-for-1","slug":"sequential-randomized-smoothing-for-1","title":"Sequential Randomized Smoothing for Adversarially Robust Speech Recognition","date":"2021-11-05","arxiv_id":"2112.03000","repositories_listed":1,"syntology":null},{"url":"/paper/a-transfer-learning-based-approach-for","slug":"a-transfer-learning-based-approach-for","title":"A transfer learning based approach for pronunciation scoring","date":"2021-11-01","arxiv_id":"2111.00976","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-augmented-transducer-networks","slug":"cross-attention-augmented-transducer-networks","title":"Cross Attention Augmented Transducer Networks for Simultaneous Translation","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-evaluation-of-language-models-for","slug":"intrinsic-evaluation-of-language-models-for","title":"Intrinsic evaluation of language models for code-switching","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revealing-and-protecting-labels-in","slug":"revealing-and-protecting-labels-in","title":"Revealing and Protecting Labels in Distributed Training","date":"2021-10-31","arxiv_id":"2111.00556","repositories_listed":1,"syntology":null},{"url":"/paper/aequevox-automated-fairness-testing-of-speech","slug":"aequevox-automated-fairness-testing-of-speech","title":"AequeVox: Automated Fairness Testing of Speech Recognition Systems","date":"2021-10-19","arxiv_id":"2110.09843","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-speaker-adaptation-approach-for-asr","slug":"a-unified-speaker-adaptation-approach-for-asr","title":"A Unified Speaker Adaptation Approach for ASR","date":"2021-10-16","arxiv_id":"2110.08545","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-introductions-in-podcast-episodes","slug":"identifying-introductions-in-podcast-episodes","title":"Identifying Introductions in Podcast Episodes from Automatically Generated Transcripts","date":"2021-10-14","arxiv_id":"2110.07096","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-conditional-end-to-end-asr-with","slug":"hierarchical-conditional-end-to-end-asr-with","title":"Hierarchical Conditional End-to-End ASR with CTC and Multi-Granular Subword Units","date":"2021-10-08","arxiv_id":"2110.04109","repositories_listed":1,"syntology":null},{"url":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","repositories_listed":1,"syntology":null},{"url":"/paper/fastcorrect-2-fast-error-correction-on","slug":"fastcorrect-2-fast-error-correction-on","title":"FastCorrect 2: Fast Error Correction on Multiple Candidates for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":"2109.14420","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fastcorrect-2-fast-error-correction-on#ran","syntology_url":"https://syntology.ai/paper/2109.14420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14420"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/factorized-neural-transducer-for-efficient","slug":"factorized-neural-transducer-for-efficient","title":"Factorized Neural Transducer for Efficient Language Model Adaptation","date":"2021-09-27","arxiv_id":"2110.01500","repositories_listed":1,"syntology":null},{"url":"/paper/fast-md-fast-multi-decoder-end-to-end-speech","slug":"fast-md-fast-multi-decoder-end-to-end-speech","title":"Fast-MD: Fast Multi-Decoder End-to-End Speech Translation with Non-Autoregressive Hidden Intermediates","date":"2021-09-27","arxiv_id":"2109.12804","repositories_listed":1,"syntology":null},{"url":"/paper/performance-efficiency-trade-offs-in","slug":"performance-efficiency-trade-offs-in","title":"Performance-Efficiency Trade-offs in Unsupervised Pre-training for Speech Recognition","date":"2021-09-14","arxiv_id":"2109.06870","repositories_listed":1,"syntology":null},{"url":"/paper/multi-sentence-resampling-a-simple-approach","slug":"multi-sentence-resampling-a-simple-approach","title":"Multi-Sentence Resampling: A Simple Approach to Alleviate Dataset Length Bias and Beam-Search Degradation","date":"2021-09-13","arxiv_id":"2109.06253","repositories_listed":1,"syntology":null},{"url":"/paper/vietnamese-end-to-end-speech-recognition","slug":"vietnamese-end-to-end-speech-recognition","title":"Vietnamese end-to-end speech recognition using wav2vec 2.0","date":"2021-09-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficient-conformer-progressive-downsampling","slug":"efficient-conformer-progressive-downsampling","title":"Efficient conformer: Progressive downsampling and grouped attention for automatic speech recognition","date":"2021-08-31","arxiv_id":"2109.01163","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-with-joint","slug":"end-to-end-speech-recognition-with-joint","title":"End-to-End Speech Recognition With Joint Dereverberation Of Sub-Band Autoregressive Envelopes","date":"2021-08-09","arxiv_id":"2108.03975","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-distillation-from-bert-transformer","slug":"knowledge-distillation-from-bert-transformer","title":"Knowledge Distillation from BERT Transformer to Speech Transformer for Intent Classification","date":"2021-08-05","arxiv_id":"2108.02598","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-multilingual-end-to-end-speech","slug":"a-study-of-multilingual-end-to-end-speech","title":"A Study of Multilingual End-to-End Speech Recognition for Kazakh, Russian, and English","date":"2021-08-03","arxiv_id":"2108.01280","repositories_listed":1,"syntology":null},{"url":"/paper/the-history-of-speech-recognition-to-the-year","slug":"the-history-of-speech-recognition-to-the-year","title":"The History of Speech Recognition to the Year 2030","date":"2021-07-30","arxiv_id":"2108.00084","repositories_listed":1,"syntology":null},{"url":"/paper/usc-an-open-source-uzbek-speech-corpus-and","slug":"usc-an-open-source-uzbek-speech-corpus-and","title":"USC: An Open-Source Uzbek Speech Corpus and Initial Speech Recognition Experiments","date":"2021-07-30","arxiv_id":"2107.14419","repositories_listed":1,"syntology":null},{"url":"/paper/brazilian-portuguese-speech-recognition-using","slug":"brazilian-portuguese-speech-recognition-using","title":"Brazilian Portuguese Speech Recognition Using Wav2vec 2.0","date":"2021-07-23","arxiv_id":"2107.11414","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-model-with-self-adaptive-sliding","slug":"sequence-model-with-self-adaptive-sliding","title":"Sequence Model with Self-Adaptive Sliding Window for Efficient Spoken Document Segmentation","date":"2021-07-20","arxiv_id":"2107.09278","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-end-to-end-asr-based-on-blockwise","slug":"streaming-end-to-end-asr-based-on-blockwise","title":"Streaming End-to-End ASR based on Blockwise Non-Autoregressive Models","date":"2021-07-20","arxiv_id":"2107.09428","repositories_listed":1,"syntology":null},{"url":"/paper/token-level-supervised-contrastive-learning","slug":"token-level-supervised-contrastive-learning","title":"Token-Level Supervised Contrastive Learning for Punctuation Restoration","date":"2021-07-19","arxiv_id":"2107.09099","repositories_listed":1,"syntology":null},{"url":"/paper/strode-stochastic-boundary-ordinary","slug":"strode-stochastic-boundary-ordinary","title":"STRODE: Stochastic Boundary Ordinary Differential Equation","date":"2021-07-17","arxiv_id":"2107.08273","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/strode-stochastic-boundary-ordinary#ran","syntology_url":"https://syntology.ai/paper/2107.08273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08273"}},"official":{"repos":["Waffle-Liu/STRODE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/layer-wise-analysis-of-a-self-supervised","slug":"layer-wise-analysis-of-a-self-supervised","title":"Layer-wise Analysis of a Self-supervised Speech Representation Model","date":"2021-07-10","arxiv_id":"2107.04734","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-ctc-crf-based-end-to-end-speech","slug":"advancing-ctc-crf-based-end-to-end-speech","title":"Advancing CTC-CRF Based End-to-End Speech Recognition with Wordpieces and Conformers","date":"2021-07-07","arxiv_id":"2107.03007","repositories_listed":1,"syntology":null},{"url":"/paper/instant-one-shot-word-learning-for-context","slug":"instant-one-shot-word-learning-for-context","title":"Instant One-Shot Word-Learning for Context-Specific Neural Sequence-to-Sequence Speech Recognition","date":"2021-07-05","arxiv_id":"2107.02268","repositories_listed":1,"syntology":null},{"url":"/paper/tenet-a-time-reversal-enhancement-network-for","slug":"tenet-a-time-reversal-enhancement-network-for","title":"TENET: A Time-reversal Enhancement Network for Noise-robust ASR","date":"2021-07-04","arxiv_id":"2107.01531","repositories_listed":1,"syntology":null},{"url":"/paper/relaxed-attention-a-simple-method-to-boost","slug":"relaxed-attention-a-simple-method-to-boost","title":"Relaxed Attention: A Simple Method to Boost Performance of End-to-End Automatic Speech Recognition","date":"2021-07-02","arxiv_id":"2107.01275","repositories_listed":1,"syntology":null},{"url":"/paper/combining-frame-synchronous-and-label","slug":"combining-frame-synchronous-and-label","title":"Combining Frame-Synchronous and Label-Synchronous Systems for Speech Recognition","date":"2021-07-01","arxiv_id":"2107.00764","repositories_listed":1,"syntology":null},{"url":"/paper/pretext-tasks-selection-for-multitask-self","slug":"pretext-tasks-selection-for-multitask-self","title":"Pretext Tasks selection for multitask self-supervised speech representation learning","date":"2021-07-01","arxiv_id":"2107.00594","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pretext-tasks-selection-for-multitask-self#ran","syntology_url":"https://syntology.ai/paper/2107.00594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00594"}},"official":{"repos":["salah-zaiem/PL-groupselection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/momentum-pseudo-labeling-for-semi-supervised","slug":"momentum-pseudo-labeling-for-semi-supervised","title":"Momentum Pseudo-Labeling for Semi-Supervised Speech Recognition","date":"2021-06-16","arxiv_id":"2106.08922","repositories_listed":1,"syntology":null},{"url":"/paper/multi-speaker-asr-combining-non","slug":"multi-speaker-asr-combining-non","title":"Multi-Speaker ASR Combining Non-Autoregressive Conformer CTC and Conditional Speaker Chain","date":"2021-06-16","arxiv_id":"2106.08595","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-use-of-prosody-in-constituency","slug":"assessing-the-use-of-prosody-in-constituency","title":"Assessing the Use of Prosody in Constituency Parsing of Imperfect Transcripts","date":"2021-06-14","arxiv_id":"2106.07794","repositories_listed":1,"syntology":null},{"url":"/paper/learning-audio-visual-dereverberation","slug":"learning-audio-visual-dereverberation","title":"Learning Audio-Visual Dereverberation","date":"2021-06-14","arxiv_id":"2106.07732","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-external-pos-tagger-for","slug":"incorporating-external-pos-tagger-for","title":"Incorporating External POS Tagger for Punctuation Restoration","date":"2021-06-12","arxiv_id":"2106.06731","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-contextual-language-model","slug":"attention-based-contextual-language-model","title":"Attention-based Contextual Language Model Adaptation for Speech Recognition","date":"2021-06-02","arxiv_id":"2106.01451","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/attention-based-contextual-language-model#ran","syntology_url":"https://syntology.ai/paper/2106.01451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01451"}},"official":{"repos":["amazon-research/contextual-attention-nlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-speech-recognition-in-sanskrit-a","slug":"automatic-speech-recognition-in-sanskrit-a","title":"Automatic Speech Recognition in Sanskrit: A New Speech Corpus and Modelling Insights","date":"2021-06-02","arxiv_id":"2106.05852","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-reordering-capability-in","slug":"investigating-the-reordering-capability-in","title":"Investigating the Reordering Capability in CTC-based Non-Autoregressive End-to-End Speech Translation","date":"2021-05-11","arxiv_id":"2105.04840","repositories_listed":1,"syntology":null},{"url":"/paper/fastcorrect-fast-error-correction-with-edit","slug":"fastcorrect-fast-error-correction-with-edit","title":"FastCorrect: Fast Error Correction with Edit Alignment for Automatic Speech Recognition","date":"2021-05-09","arxiv_id":"2105.03842","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/fastcorrect-fast-error-correction-with-edit#ran","syntology_url":"https://syntology.ai/paper/2105.03842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03842"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/end-to-end-speech-recognition-from-federated","slug":"end-to-end-speech-recognition-from-federated","title":"End-to-End Speech Recognition from Federated Acoustic Models","date":"2021-04-29","arxiv_id":"2104.14297","repositories_listed":1,"syntology":null},{"url":"/paper/lebenchmark-a-reproducible-framework-for","slug":"lebenchmark-a-reproducible-framework-for","title":"LeBenchmark: A Reproducible Framework for Assessing Self-Supervised Representation Learning from Speech","date":"2021-04-23","arxiv_id":"2104.11462","repositories_listed":1,"syntology":null},{"url":"/paper/a-method-to-reveal-speaker-identity-in","slug":"a-method-to-reveal-speaker-identity-in","title":"A Method to Reveal Speaker Identity in Distributed ASR Training, and How to Counter It","date":"2021-04-15","arxiv_id":"2104.07815","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-independence-for-pretext-task","slug":"conditional-independence-for-pretext-task","title":"Conditional independence for pretext task selection in Self-supervised speech representation learning","date":"2021-04-15","arxiv_id":"2104.07388","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-speech-recognition-with","slug":"cross-domain-speech-recognition-with","title":"Cross-domain Speech Recognition with Unsupervised Character-level Distribution Matching","date":"2021-04-15","arxiv_id":"2104.07491","repositories_listed":1,"syntology":null},{"url":"/paper/nemo-inverse-text-normalization-from","slug":"nemo-inverse-text-normalization-from","title":"NeMo Inverse Text Normalization: From Development To Production","date":"2021-04-11","arxiv_id":"2104.05055","repositories_listed":1,"syntology":null},{"url":"/paper/nemo-toolbox-for-speech-dataset-construction","slug":"nemo-toolbox-for-speech-dataset-construction","title":"A Toolbox for Construction and Analysis of Speech Datasets","date":"2021-04-11","arxiv_id":"2104.04896","repositories_listed":1,"syntology":null},{"url":"/paper/rnn-transducer-models-for-spoken-language","slug":"rnn-transducer-models-for-spoken-language","title":"RNN Transducer Models For Spoken Language Understanding","date":"2021-04-08","arxiv_id":"2104.03842","repositories_listed":1,"syntology":null},{"url":"/paper/speak-or-chat-with-me-end-to-end-spoken","slug":"speak-or-chat-with-me-end-to-end-spoken","title":"Speak or Chat with Me: End-to-End Spoken Language Understanding System with Flexible Inputs","date":"2021-04-07","arxiv_id":"2104.05752","repositories_listed":1,"syntology":null},{"url":"/paper/lt-lm-a-novel-non-autoregressive-language","slug":"lt-lm-a-novel-non-autoregressive-language","title":"LT-LM: a novel non-autoregressive language model for single-shot lattice rescoring","date":"2021-04-06","arxiv_id":"2104.02526","repositories_listed":1,"syntology":null},{"url":"/paper/exkaldi-rt-a-real-time-automatic-speech","slug":"exkaldi-rt-a-real-time-automatic-speech","title":"ExKaldi-RT: A Real-Time Automatic Speech Recognition Extension Toolkit of Kaldi","date":"2021-04-03","arxiv_id":"2104.01384","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-fly-aligned-data-augmentation-for","slug":"on-the-fly-aligned-data-augmentation-for","title":"On-the-Fly Aligned Data Augmentation for Sequence-to-Sequence ASR","date":"2021-04-03","arxiv_id":"2104.01393","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-and-code-switching-asr","slug":"multilingual-and-code-switching-asr","title":"Multilingual and code-switching ASR challenges for low resource Indian languages","date":"2021-04-01","arxiv_id":"2104.00235","repositories_listed":1,"syntology":null},{"url":"/paper/q-asr-integer-only-zero-shot-quantization-for","slug":"q-asr-integer-only-zero-shot-quantization-for","title":"Integer-only Zero-shot Quantization for Efficient Speech Recognition","date":"2021-03-31","arxiv_id":"2103.16827","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-bias-in-automatic-speech","slug":"quantifying-bias-in-automatic-speech","title":"Quantifying Bias in Automatic Speech Recognition","date":"2021-03-28","arxiv_id":"2103.15122","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-neural-representations-for","slug":"leveraging-neural-representations-for","title":"Leveraging pre-trained representations to improve access to untranscribed speech from endangered languages","date":"2021-03-26","arxiv_id":"2103.14583","repositories_listed":1,"syntology":null},{"url":"/paper/sok-a-modularized-approach-to-study-the","slug":"sok-a-modularized-approach-to-study-the","title":"SoK: A Modularized Approach to Study the Security of Automatic Speech Recognition Systems","date":"2021-03-19","arxiv_id":"2103.10651","repositories_listed":1,"syntology":null},{"url":"/paper/fast-development-of-asr-in-african-languages","slug":"fast-development-of-asr-in-african-languages","title":"Fast Development of ASR in African Languages using Self Supervised Speech Representation Learning","date":"2021-03-16","arxiv_id":"2103.08993","repositories_listed":1,"syntology":null},{"url":"/paper/a-parallelizable-lattice-rescoring-strategy","slug":"a-parallelizable-lattice-rescoring-strategy","title":"A Parallelizable Lattice Rescoring Strategy with Neural Language Models","date":"2021-03-08","arxiv_id":"2103.05081","repositories_listed":1,"syntology":null},{"url":"/paper/waveguard-understanding-and-mitigating-audio","slug":"waveguard-understanding-and-mitigating-audio","title":"WaveGuard: Understanding and Mitigating Audio Adversarial Examples","date":"2021-03-04","arxiv_id":"2103.03344","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/waveguard-understanding-and-mitigating-audio#ran","syntology_url":"https://syntology.ai/paper/2103.03344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.03344"}},"official":null}},{"url":"/paper/exploiting-attention-based-sequence-to","slug":"exploiting-attention-based-sequence-to","title":"Exploiting Attention-based Sequence-to-Sequence Architectures for Sound Event Localization","date":"2021-02-28","arxiv_id":"2103.00417","repositories_listed":1,"syntology":null},{"url":"/paper/data-fusion-for-audiovisual-speaker","slug":"data-fusion-for-audiovisual-speaker","title":"Data Fusion for Audiovisual Speaker Localization: Extending Dynamic Stream Weights to the Spatial Domain","date":"2021-02-23","arxiv_id":"2102.11588","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-phonetic-neural-model-for-correction","slug":"hybrid-phonetic-neural-model-for-correction","title":"Hybrid phonetic-neural model for correction in speech recognition systems","date":"2021-02-12","arxiv_id":"2102.06744","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-language-models-with-lstm-based","slug":"transformer-language-models-with-lstm-based","title":"Transformer Language Models with LSTM-based Cross-utterance Information Representation","date":"2021-02-12","arxiv_id":"2102.06474","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-end-to-end-models-for","slug":"an-investigation-of-end-to-end-models-for","title":"An Investigation of End-to-End Models for Robust Speech Recognition","date":"2021-02-11","arxiv_id":"2102.06237","repositories_listed":1,"syntology":null},{"url":"/paper/dompteur-taming-audio-adversarial-examples","slug":"dompteur-taming-audio-adversarial-examples","title":"Dompteur: Taming Audio Adversarial Examples","date":"2021-02-10","arxiv_id":"2102.05431","repositories_listed":1,"syntology":null},{"url":"/paper/effects-of-layer-freezing-when-transferring","slug":"effects-of-layer-freezing-when-transferring","title":"Effects of Layer Freezing on Transferring a Speech Recognition System to Under-resourced Languages","date":"2021-02-08","arxiv_id":"2102.04097","repositories_listed":1,"syntology":null},{"url":"/paper/confusion2vec-2-0-enriching-ambiguous-spoken","slug":"confusion2vec-2-0-enriching-ambiguous-spoken","title":"Confusion2vec 2.0: Enriching Ambiguous Spoken Language Representations with Subwords","date":"2021-02-03","arxiv_id":"2102.02270","repositories_listed":1,"syntology":null},{"url":"/paper/bendr-using-transformers-and-a-contrastive","slug":"bendr-using-transformers-and-a-contrastive","title":"BENDR: using transformers and a contrastive self-supervised learning task to learn from massive amounts of EEG data","date":"2021-01-28","arxiv_id":"2101.12037","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-speech-recognition-by-end-to-end","slug":"arabic-speech-recognition-by-end-to-end","title":"Arabic Speech Recognition by End-to-End, Modular Systems and Human","date":"2021-01-21","arxiv_id":"2101.08454","repositories_listed":1,"syntology":null},{"url":"/paper/av-taris-online-audio-visual-speech","slug":"av-taris-online-audio-visual-speech","title":"AV Taris: Online Audio-Visual Speech Recognition","date":"2020-12-14","arxiv_id":"2012.07467","repositories_listed":1,"syntology":null}],"record_sha256":"edff3776fa089e0b09edf2d7860124e7b6c9fbd348e88e3e21d14122b2fc6e0b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}