{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/9","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":58,"rows_per_page":100,"rows":[801,900],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/8","next":"/task/speech-recognition-1/papers/10","papers":[{"url":"/paper/flute-a-scalable-extensible-framework-for","slug":"flute-a-scalable-extensible-framework-for","title":"FLUTE: A Scalable, Extensible Framework for High-Performance Federated Learning Simulations","date":"2022-03-25","arxiv_id":"2203.13789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flute-a-scalable-extensible-framework-for#ran","syntology_url":"https://syntology.ai/paper/2203.13789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13789"}},"official":{"repos":["microsoft/msrflute"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-enhanced-and-noise-aware-networks-for","slug":"speech-enhanced-and-noise-aware-networks-for","title":"Speech-enhanced and Noise-aware Networks for Robust Speech Recognition","date":"2022-03-25","arxiv_id":"2203.13696","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-for-speech","slug":"automatic-speech-recognition-for-speech","title":"Automatic Speech Recognition for Speech Assessment of Persian Preschool Children","date":"2022-03-24","arxiv_id":"2203.12886","repositories_listed":1,"syntology":null},{"url":"/paper/neural-predictor-for-black-box-adversarial","slug":"neural-predictor-for-black-box-adversarial","title":"Neural Predictor for Black-Box Adversarial Attacks on Speech Recognition","date":"2022-03-18","arxiv_id":"2203.09849","repositories_listed":1,"syntology":null},{"url":"/paper/modelling-word-learning-and-recognition-using","slug":"modelling-word-learning-and-recognition-using","title":"Modelling word learning and recognition using visually grounded speech","date":"2022-03-14","arxiv_id":"2203.06937","repositories_listed":1,"syntology":null},{"url":"/paper/red-ace-robust-error-detection-for-asr-using-1","slug":"red-ace-robust-error-detection-for-asr-using-1","title":"RED-ACE: Robust Error Detection for ASR using Confidence Embeddings","date":"2022-03-14","arxiv_id":"2203.07172","repositories_listed":1,"syntology":null},{"url":"/paper/dual-textless-spoken-question-answering-with-1","slug":"dual-textless-spoken-question-answering-with-1","title":"DUAL: Discrete Spoken Unit Adaptive Learning for Textless Spoken Question Answering","date":"2022-03-09","arxiv_id":"2203.04911","repositories_listed":1,"syntology":null},{"url":"/paper/creating-speech-to-speech-corpus-from-dubbed","slug":"creating-speech-to-speech-corpus-from-dubbed","title":"Creating Speech-to-Speech Corpus from Dubbed Series","date":"2022-03-07","arxiv_id":"2203.03601","repositories_listed":1,"syntology":null},{"url":"/paper/towards-contextual-spelling-correction-for","slug":"towards-contextual-spelling-correction-for","title":"Towards Contextual Spelling Correction for Customization of End-to-end Speech Recognition Systems","date":"2022-03-02","arxiv_id":"2203.00888","repositories_listed":1,"syntology":null},{"url":"/paper/sentiment-word-aware-multimodal-refinement","slug":"sentiment-word-aware-multimodal-refinement","title":"Sentiment Word Aware Multimodal Refinement for Multimodal Sentiment Analysis with ASR Errors","date":"2022-03-01","arxiv_id":"2203.00257","repositories_listed":1,"syntology":null},{"url":"/paper/icassp-2022-acoustic-echo-cancellation","slug":"icassp-2022-acoustic-echo-cancellation","title":"ICASSP 2022 Acoustic Echo Cancellation Challenge","date":"2022-02-27","arxiv_id":"2202.13290","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-uni-modal-self-supervised-learning-1","slug":"leveraging-uni-modal-self-supervised-learning-1","title":"Leveraging Unimodal Self-Supervised Learning for Multimodal Audio-Visual Speech Recognition","date":"2022-02-24","arxiv_id":"2203.07996","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/leveraging-uni-modal-self-supervised-learning-1#ran","syntology_url":"https://syntology.ai/paper/2203.07996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07996"}},"official":{"repos":["lumia-group/leveraging-self-supervised-learning-for-avsr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flowsense-monitoring-airflow-in-building","slug":"flowsense-monitoring-airflow-in-building","title":"FlowSense: Monitoring Airflow in Building Ventilation Systems Using Audio Sensing","date":"2022-02-22","arxiv_id":"2202.11136","repositories_listed":1,"syntology":null},{"url":"/paper/improving-ctc-based-speech-recognition-via","slug":"improving-ctc-based-speech-recognition-via","title":"Improving CTC-based speech recognition via knowledge transferring from pre-trained language models","date":"2022-02-22","arxiv_id":"2203.03582","repositories_listed":1,"syntology":null},{"url":"/paper/spanish-and-english-phoneme-recognition-by","slug":"spanish-and-english-phoneme-recognition-by","title":"Spanish and English Phoneme Recognition by Training on Simulated Classroom Audio Recordings of Collaborative Learning Environments","date":"2022-02-21","arxiv_id":"2202.10536","repositories_listed":1,"syntology":null},{"url":"/paper/aishell-ner-named-entity-recognition-from","slug":"aishell-ner-named-entity-recognition-from","title":"AISHELL-NER: Named Entity Recognition from Chinese Speech","date":"2022-02-17","arxiv_id":"2202.08533","repositories_listed":1,"syntology":null},{"url":"/paper/adima-abuse-detection-in-multilingual-audio","slug":"adima-abuse-detection-in-multilingual-audio","title":"ADIMA: Abuse Detection In Multilingual Audio","date":"2022-02-16","arxiv_id":"2202.07991","repositories_listed":1,"syntology":null},{"url":"/paper/improving-automatic-speech-recognition-for","slug":"improving-automatic-speech-recognition-for","title":"Improving Automatic Speech Recognition for Non-Native English with Transfer Learning and Language Model Decoding","date":"2022-02-10","arxiv_id":"2202.05209","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-adapter-transfer-of-self-supervised","slug":"efficient-adapter-transfer-of-self-supervised","title":"Efficient Adapter Transfer of Self-Supervised Speech Models for Automatic Speech Recognition","date":"2022-02-07","arxiv_id":"2202.03218","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-adapter-transfer-of-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2202.03218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03218"}},"official":null}},{"url":"/paper/streaming-multi-talker-asr-with-token-level","slug":"streaming-multi-talker-asr-with-token-level","title":"Streaming Multi-Talker ASR with Token-Level Serialized Output Training","date":"2022-02-02","arxiv_id":"2202.00842","repositories_listed":1,"syntology":null},{"url":"/paper/nas-bench-suite-nas-evaluation-is-now-1","slug":"nas-bench-suite-nas-evaluation-is-now-1","title":"NAS-Bench-Suite: NAS Evaluation is (Now) Surprisingly Easy","date":"2022-01-31","arxiv_id":"2201.13396","repositories_listed":1,"syntology":null},{"url":"/paper/improving-end-to-end-contextual-speech","slug":"improving-end-to-end-contextual-speech","title":"Improving End-to-End Contextual Speech Recognition with Fine-Grained Contextual Knowledge Selection","date":"2022-01-30","arxiv_id":"2201.12806","repositories_listed":1,"syntology":null},{"url":"/paper/star-temporal-classification-sequence","slug":"star-temporal-classification-sequence","title":"Star Temporal Classification: Sequence Classification with Partially Labeled Data","date":"2022-01-28","arxiv_id":"2201.12208","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-phonetic-inventories-with","slug":"discovering-phonetic-inventories-with","title":"Discovering Phonetic Inventories with Crosslingual Automatic Speech Recognition","date":"2022-01-26","arxiv_id":"2201.11207","repositories_listed":1,"syntology":null},{"url":"/paper/unified-multimodal-punctuation-restoration","slug":"unified-multimodal-punctuation-restoration","title":"Unified Multimodal Punctuation Restoration Framework for Mixed-Modality Corpus","date":"2022-01-24","arxiv_id":"2202.00468","repositories_listed":1,"syntology":null},{"url":"/paper/ci-avsr-a-cantonese-audio-visual-speech","slug":"ci-avsr-a-cantonese-audio-visual-speech","title":"CI-AVSR: A Cantonese Audio-Visual Speech Dataset for In-car Command Recognition","date":"2022-01-11","arxiv_id":"2201.03804","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ci-avsr-a-cantonese-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2201.03804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.03804"}},"official":{"repos":["hltchkust/ci-avsr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-architecture-search-for-lf-mmi-trained","slug":"neural-architecture-search-for-lf-mmi-trained","title":"Neural Architecture Search For LF-MMI Trained Time Delay Neural Networks","date":"2022-01-08","arxiv_id":"2201.03943","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-datasets-in","slug":"automatic-speech-recognition-datasets-in","title":"Automatic Speech Recognition Datasets in Cantonese: A Survey and New Dataset","date":"2022-01-07","arxiv_id":"2201.02419","repositories_listed":1,"syntology":null},{"url":"/paper/improving-mandarin-end-to-end-speech","slug":"improving-mandarin-end-to-end-speech","title":"Improving Mandarin End-to-End Speech Recognition with Word N-gram Language Model","date":"2022-01-06","arxiv_id":"2201.01995","repositories_listed":1,"syntology":null},{"url":"/paper/robust-self-supervised-audio-visual-speech","slug":"robust-self-supervised-audio-visual-speech","title":"Robust Self-Supervised Audio-Visual Speech Recognition","date":"2022-01-05","arxiv_id":"2201.01763","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-self-supervised-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2201.01763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01763"}},"official":{"repos":["facebookresearch/av_hubert"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/a-hierarchical-model-for-spoken-language","slug":"a-hierarchical-model-for-spoken-language","title":"A Discriminative Hierarchical PLDA-based Model for Spoken Language Recognition","date":"2022-01-04","arxiv_id":"2201.01364","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-attacks-against-windows-pe","slug":"adversarial-attacks-against-windows-pe","title":"Adversarial Attacks against Windows PE Malware Detection: A Survey of the State-of-the-Art","date":"2021-12-23","arxiv_id":"2112.12310","repositories_listed":1,"syntology":null},{"url":"/paper/regularizing-end-to-end-speech-translation","slug":"regularizing-end-to-end-speech-translation","title":"Regularizing End-to-End Speech Translation with Triangular Decomposition Agreement","date":"2021-12-21","arxiv_id":"2112.10991","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-for-monolingual-end-to-end","slug":"continual-learning-for-monolingual-end-to-end","title":"Continual Learning for Monolingual End-to-End Automatic Speech Recognition","date":"2021-12-17","arxiv_id":"2112.09427","repositories_listed":1,"syntology":null},{"url":"/paper/automated-deep-learning-neural-architecture","slug":"automated-deep-learning-neural-architecture","title":"Automated Deep Learning: Neural Architecture Search Is Not the End","date":"2021-12-16","arxiv_id":"2112.09245","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automated-deep-learning-neural-architecture#ran","syntology_url":"https://syntology.ai/paper/2112.09245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.09245"}},"official":{"repos":["D-X-Y/Awesome-AutoDL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-learning-for-speech-1","slug":"self-supervised-learning-for-speech-1","title":"Self-Supervised Learning for speech recognition with Intermediate layer supervision","date":"2021-12-16","arxiv_id":"2112.08778","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-use-of-external-data-for-spoken-named","slug":"on-the-use-of-external-data-for-spoken-named","title":"On the Use of External Data for Spoken Named Entity Recognition","date":"2021-12-14","arxiv_id":"2112.07648","repositories_listed":1,"syntology":null},{"url":"/paper/x-vector-based-voice-activity-detection-for-1","slug":"x-vector-based-voice-activity-detection-for-1","title":"X-Vector based voice activity detection for multi-genre broadcast speech-to-text","date":"2021-12-09","arxiv_id":"2112.05016","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-training-and-decoding-for-end-to","slug":"consistent-training-and-decoding-for-end-to","title":"Consistent Training and Decoding For End-to-end Speech Recognition Using Lattice-free MMI","date":"2021-12-05","arxiv_id":"2112.02498","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-adaptive-multiscale-temporal","slug":"understanding-adaptive-multiscale-temporal","title":"Understanding Adaptive, Multiscale Temporal Integration In Deep Speech Recognition Systems","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/romanian-speech-recognition-experiments-from","slug":"romanian-speech-recognition-experiments-from","title":"Romanian Speech Recognition Experiments from the ROBIN Project","date":"2021-11-23","arxiv_id":"2111.12028","repositories_listed":1,"syntology":null},{"url":"/paper/slue-new-benchmark-tasks-for-spoken-language","slug":"slue-new-benchmark-tasks-for-spoken-language","title":"SLUE: New Benchmark Tasks for Spoken Language Understanding Evaluation on Natural Speech","date":"2021-11-19","arxiv_id":"2111.10367","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slue-new-benchmark-tasks-for-spoken-language#ran","syntology_url":"https://syntology.ai/paper/2111.10367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10367"}},"official":{"repos":["asappresearch/slue-toolkit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/integrated-semantic-and-phonetic-post-1","slug":"integrated-semantic-and-phonetic-post-1","title":"Integrated Semantic and Phonetic Post-correction for Chinese Speech Recognition","date":"2021-11-16","arxiv_id":"2111.08400","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-semantic-driven-phoneme","slug":"self-supervised-semantic-driven-phoneme","title":"Self-supervised Semantic-driven Phoneme Discovery for Zero-resource Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-contribution-of-multiple-model","slug":"measuring-the-contribution-of-multiple-model","title":"Measuring the Contribution of Multiple Model Representations in Detecting Adversarial Instances","date":"2021-11-13","arxiv_id":"2111.07035","repositories_listed":1,"syntology":null},{"url":"/paper/asl-trigger-recognition-in-mixed-activity","slug":"asl-trigger-recognition-in-mixed-activity","title":"ASL Trigger Recognition in Mixed Activity/Signing Sequences for RF Sensor-Based User Interfaces","date":"2021-11-10","arxiv_id":"2111.05480","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-randomized-smoothing-for-1","slug":"sequential-randomized-smoothing-for-1","title":"Sequential Randomized Smoothing for Adversarially Robust Speech Recognition","date":"2021-11-05","arxiv_id":"2112.03000","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-to-sequence-modeling-for-action-1","slug":"sequence-to-sequence-modeling-for-action-1","title":"Sequence-to-Sequence Modeling for Action Identification at High Temporal Resolution","date":"2021-11-03","arxiv_id":"2111.02521","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sequence-to-sequence-modeling-for-action-1#ran","syntology_url":"https://syntology.ai/paper/2111.02521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02521"}},"official":null}},{"url":"/paper/cross-lingual-transfer-for-speech-processing","slug":"cross-lingual-transfer-for-speech-processing","title":"Cross-lingual Transfer for Speech Processing using Acoustic Language Similarity","date":"2021-11-02","arxiv_id":"2111.01326","repositories_listed":1,"syntology":null},{"url":"/paper/a-transfer-learning-based-approach-for","slug":"a-transfer-learning-based-approach-for","title":"A transfer learning based approach for pronunciation scoring","date":"2021-11-01","arxiv_id":"2111.00976","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-augmented-transducer-networks","slug":"cross-attention-augmented-transducer-networks","title":"Cross Attention Augmented Transducer Networks for Simultaneous Translation","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-robustness-of-you-only-hear-once","slug":"evaluating-robustness-of-you-only-hear-once","title":"Evaluating robustness of You Only Hear Once(YOHO) Algorithm on noisy audios in the VOICe Dataset","date":"2021-11-01","arxiv_id":"2111.01205","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-evaluation-of-language-models-for","slug":"intrinsic-evaluation-of-language-models-for","title":"Intrinsic evaluation of language models for code-switching","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/orthographic-transliteration-for-kabyle","slug":"orthographic-transliteration-for-kabyle","title":"Orthographic Transliteration for Kabyle Speech Recognition","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revealing-and-protecting-labels-in","slug":"revealing-and-protecting-labels-in","title":"Revealing and Protecting Labels in Distributed Training","date":"2021-10-31","arxiv_id":"2111.00556","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-joint-decoding-based-multi-talker","slug":"revisiting-joint-decoding-based-multi-talker","title":"Revisiting joint decoding based multi-talker speech recognition with DNN acoustic model","date":"2021-10-31","arxiv_id":"2111.00009","repositories_listed":1,"syntology":null},{"url":"/paper/game-of-gradients-mitigating-irrelevant","slug":"game-of-gradients-mitigating-irrelevant","title":"Game of Gradients: Mitigating Irrelevant Clients in Federated Learning","date":"2021-10-23","arxiv_id":"2110.12257","repositories_listed":1,"syntology":null},{"url":"/paper/aequevox-automated-fairness-testing-of-speech","slug":"aequevox-automated-fairness-testing-of-speech","title":"AequeVox: Automated Fairness Testing of Speech Recognition Systems","date":"2021-10-19","arxiv_id":"2110.09843","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-of-french-phonetic-idiosyncrasies","slug":"analysis-of-french-phonetic-idiosyncrasies","title":"Analysis of French Phonetic Idiosyncrasies for Accent Recognition","date":"2021-10-18","arxiv_id":"2110.09179","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-speaker-adaptation-approach-for-asr","slug":"a-unified-speaker-adaptation-approach-for-asr","title":"A Unified Speaker Adaptation Approach for ASR","date":"2021-10-16","arxiv_id":"2110.08545","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-introductions-in-podcast-episodes","slug":"identifying-introductions-in-podcast-episodes","title":"Identifying Introductions in Podcast Episodes from Automatically Generated Transcripts","date":"2021-10-14","arxiv_id":"2110.07096","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-wav2vec-2-0-fine-tuning-for","slug":"exploring-wav2vec-2-0-fine-tuning-for","title":"Exploring Wav2vec 2.0 fine-tuning for improved speech emotion recognition","date":"2021-10-12","arxiv_id":"2110.06309","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-wav2vec-2-0-fine-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2110.06309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06309"}},"official":{"repos":["b04901014/FT-w2v2-ser"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/long-expressive-memory-for-sequence-modeling-1","slug":"long-expressive-memory-for-sequence-modeling-1","title":"Long Expressive Memory for Sequence Modeling","date":"2021-10-10","arxiv_id":"2110.04744","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":1,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/long-expressive-memory-for-sequence-modeling-1#ran","syntology_url":"https://syntology.ai/paper/2110.04744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04744"}},"official":{"repos":["tk-rusch/lem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/arabic-speech-emotion-recognition-employing","slug":"arabic-speech-emotion-recognition-employing","title":"Arabic Speech Emotion Recognition Employing Wav2vec2.0 and HuBERT Based on BAVED Dataset","date":"2021-10-09","arxiv_id":"2110.04425","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-conditional-end-to-end-asr-with","slug":"hierarchical-conditional-end-to-end-asr-with","title":"Hierarchical Conditional End-to-End ASR with CTC and Multi-Granular Subword Units","date":"2021-10-08","arxiv_id":"2110.04109","repositories_listed":1,"syntology":null},{"url":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","repositories_listed":1,"syntology":null},{"url":"/paper/late-reverberation-suppression-using-u-nets","slug":"late-reverberation-suppression-using-u-nets","title":"Late reverberation suppression using U-nets","date":"2021-10-05","arxiv_id":"2110.02144","repositories_listed":1,"syntology":null},{"url":"/paper/federated-learning-in-asr-not-as-easy-as-you","slug":"federated-learning-in-asr-not-as-easy-as-you","title":"Federated Learning in ASR: Not as Easy as You Think","date":"2021-09-30","arxiv_id":"2109.15108","repositories_listed":1,"syntology":null},{"url":"/paper/fastcorrect-2-fast-error-correction-on","slug":"fastcorrect-2-fast-error-correction-on","title":"FastCorrect 2: Fast Error Correction on Multiple Candidates for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":"2109.14420","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fastcorrect-2-fast-error-correction-on#ran","syntology_url":"https://syntology.ai/paper/2109.14420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14420"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-robust-r-u-evaluating-task-oriented","slug":"how-robust-r-u-evaluating-task-oriented","title":"\"How Robust r u?\": Evaluating Task-Oriented Dialogue Systems on Spoken Conversations","date":"2021-09-28","arxiv_id":"2109.13489","repositories_listed":1,"syntology":null},{"url":"/paper/factorized-neural-transducer-for-efficient","slug":"factorized-neural-transducer-for-efficient","title":"Factorized Neural Transducer for Efficient Language Model Adaptation","date":"2021-09-27","arxiv_id":"2110.01500","repositories_listed":1,"syntology":null},{"url":"/paper/fast-md-fast-multi-decoder-end-to-end-speech","slug":"fast-md-fast-multi-decoder-end-to-end-speech","title":"Fast-MD: Fast Multi-Decoder End-to-End Speech Translation with Non-Autoregressive Hidden Intermediates","date":"2021-09-27","arxiv_id":"2109.12804","repositories_listed":1,"syntology":null},{"url":"/paper/sd-qa-spoken-dialectal-question-answering-for","slug":"sd-qa-spoken-dialectal-question-answering-for","title":"SD-QA: Spoken Dialectal Question Answering for the Real World","date":"2021-09-24","arxiv_id":"2109.12072","repositories_listed":1,"syntology":{"n":22,"n_ran":20,"n_constructed":0,"n_ran_checked":20,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":0,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sd-qa-spoken-dialectal-question-answering-for#ran","syntology_url":"https://syntology.ai/paper/2109.12072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.12072"}},"official":{"repos":["ffaisal93/sd-qa"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":20,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robustness-analysis-of-deep-learning","slug":"robustness-analysis-of-deep-learning","title":"Robustness Analysis of Deep Learning Frameworks on Mobile Platforms","date":"2021-09-20","arxiv_id":"2109.09869","repositories_listed":1,"syntology":null},{"url":"/paper/ai-accelerator-survey-and-trends","slug":"ai-accelerator-survey-and-trends","title":"AI Accelerator Survey and Trends","date":"2021-09-18","arxiv_id":"2109.08957","repositories_listed":1,"syntology":null},{"url":"/paper/performance-efficiency-trade-offs-in","slug":"performance-efficiency-trade-offs-in","title":"Performance-Efficiency Trade-offs in Unsupervised Pre-training for Speech Recognition","date":"2021-09-14","arxiv_id":"2109.06870","repositories_listed":1,"syntology":null},{"url":"/paper/multi-sentence-resampling-a-simple-approach","slug":"multi-sentence-resampling-a-simple-approach","title":"Multi-Sentence Resampling: A Simple Approach to Alleviate Dataset Length Bias and Beam-Search Degradation","date":"2021-09-13","arxiv_id":"2109.06253","repositories_listed":1,"syntology":null},{"url":"/paper/complementing-handcrafted-features-with-raw","slug":"complementing-handcrafted-features-with-raw","title":"Complementing Handcrafted Features with Raw Waveform Using a Light-weight Auxiliary Model","date":"2021-09-06","arxiv_id":"2109.02773","repositories_listed":1,"syntology":null},{"url":"/paper/crypten-secure-multi-party-computation-meets","slug":"crypten-secure-multi-party-computation-meets","title":"CrypTen: Secure Multi-Party Computation Meets Machine Learning","date":"2021-09-02","arxiv_id":"2109.00984","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/crypten-secure-multi-party-computation-meets#ran","syntology_url":"https://syntology.ai/paper/2109.00984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00984"}},"official":null}},{"url":"/paper/vietnamese-end-to-end-speech-recognition","slug":"vietnamese-end-to-end-speech-recognition","title":"Vietnamese end-to-end speech recognition using wav2vec 2.0","date":"2021-09-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/you-only-hear-once-a-yolo-like-algorithm-for","slug":"you-only-hear-once-a-yolo-like-algorithm-for","title":"You Only Hear Once: A YOLO-like Algorithm for Audio Segmentation and Sound Event Detection","date":"2021-09-01","arxiv_id":"2109.00962","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-conformer-progressive-downsampling","slug":"efficient-conformer-progressive-downsampling","title":"Efficient conformer: Progressive downsampling and grouped attention for automatic speech recognition","date":"2021-08-31","arxiv_id":"2109.01163","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-with-joint","slug":"end-to-end-speech-recognition-with-joint","title":"End-to-End Speech Recognition With Joint Dereverberation Of Sub-Band Autoregressive Envelopes","date":"2021-08-09","arxiv_id":"2108.03975","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-distillation-from-bert-transformer","slug":"knowledge-distillation-from-bert-transformer","title":"Knowledge Distillation from BERT Transformer to Speech Transformer for Intent Classification","date":"2021-08-05","arxiv_id":"2108.02598","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-multilingual-end-to-end-speech","slug":"a-study-of-multilingual-end-to-end-speech","title":"A Study of Multilingual End-to-End Speech Recognition for Kazakh, Russian, and English","date":"2021-08-03","arxiv_id":"2108.01280","repositories_listed":1,"syntology":null},{"url":"/paper/the-history-of-speech-recognition-to-the-year","slug":"the-history-of-speech-recognition-to-the-year","title":"The History of Speech Recognition to the Year 2030","date":"2021-07-30","arxiv_id":"2108.00084","repositories_listed":1,"syntology":null},{"url":"/paper/usc-an-open-source-uzbek-speech-corpus-and","slug":"usc-an-open-source-uzbek-speech-corpus-and","title":"USC: An Open-Source Uzbek Speech Corpus and Initial Speech Recognition Experiments","date":"2021-07-30","arxiv_id":"2107.14419","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-allophone-graphs-for-language","slug":"differentiable-allophone-graphs-for-language","title":"Differentiable Allophone Graphs for Language-Universal Speech Recognition","date":"2021-07-24","arxiv_id":"2107.11628","repositories_listed":1,"syntology":null},{"url":"/paper/brazilian-portuguese-speech-recognition-using","slug":"brazilian-portuguese-speech-recognition-using","title":"Brazilian Portuguese Speech Recognition Using Wav2vec 2.0","date":"2021-07-23","arxiv_id":"2107.11414","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-model-with-self-adaptive-sliding","slug":"sequence-model-with-self-adaptive-sliding","title":"Sequence Model with Self-Adaptive Sliding Window for Efficient Spoken Document Segmentation","date":"2021-07-20","arxiv_id":"2107.09278","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-end-to-end-asr-based-on-blockwise","slug":"streaming-end-to-end-asr-based-on-blockwise","title":"Streaming End-to-End ASR based on Blockwise Non-Autoregressive Models","date":"2021-07-20","arxiv_id":"2107.09428","repositories_listed":1,"syntology":null},{"url":"/paper/token-level-supervised-contrastive-learning","slug":"token-level-supervised-contrastive-learning","title":"Token-Level Supervised Contrastive Learning for Punctuation Restoration","date":"2021-07-19","arxiv_id":"2107.09099","repositories_listed":1,"syntology":null},{"url":"/paper/strode-stochastic-boundary-ordinary","slug":"strode-stochastic-boundary-ordinary","title":"STRODE: Stochastic Boundary Ordinary Differential Equation","date":"2021-07-17","arxiv_id":"2107.08273","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/strode-stochastic-boundary-ordinary#ran","syntology_url":"https://syntology.ai/paper/2107.08273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08273"}},"official":{"repos":["Waffle-Liu/STRODE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multilingual-and-crosslingual-speech","slug":"multilingual-and-crosslingual-speech","title":"Multilingual and crosslingual speech recognition using phonological-vector based phone embeddings","date":"2021-07-11","arxiv_id":"2107.05038","repositories_listed":1,"syntology":null},{"url":"/paper/layer-wise-analysis-of-a-self-supervised","slug":"layer-wise-analysis-of-a-self-supervised","title":"Layer-wise Analysis of a Self-supervised Speech Representation Model","date":"2021-07-10","arxiv_id":"2107.04734","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-ctc-crf-based-end-to-end-speech","slug":"advancing-ctc-crf-based-end-to-end-speech","title":"Advancing CTC-CRF Based End-to-End Speech Recognition with Wordpieces and Conformers","date":"2021-07-07","arxiv_id":"2107.03007","repositories_listed":1,"syntology":null},{"url":"/paper/kosp2e-korean-speech-to-english-translation","slug":"kosp2e-korean-speech-to-english-translation","title":"Kosp2e: Korean Speech to English Translation Corpus","date":"2021-07-06","arxiv_id":"2107.02875","repositories_listed":1,"syntology":null},{"url":"/paper/instant-one-shot-word-learning-for-context","slug":"instant-one-shot-word-learning-for-context","title":"Instant One-Shot Word-Learning for Context-Specific Neural Sequence-to-Sequence Speech Recognition","date":"2021-07-05","arxiv_id":"2107.02268","repositories_listed":1,"syntology":null},{"url":"/paper/tenet-a-time-reversal-enhancement-network-for","slug":"tenet-a-time-reversal-enhancement-network-for","title":"TENET: A Time-reversal Enhancement Network for Noise-robust ASR","date":"2021-07-04","arxiv_id":"2107.01531","repositories_listed":1,"syntology":null},{"url":"/paper/relaxed-attention-a-simple-method-to-boost","slug":"relaxed-attention-a-simple-method-to-boost","title":"Relaxed Attention: A Simple Method to Boost Performance of End-to-End Automatic Speech Recognition","date":"2021-07-02","arxiv_id":"2107.01275","repositories_listed":1,"syntology":null}],"record_sha256":"c1a1da07184fde9b9c4228cb9b11f13aae89d978011e4bb526342dcb5bc1459e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}