{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/10","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":65,"rows_per_page":100,"rows":[901,1000],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/9","next":"/task/speech-recognition/papers/11","papers":[{"url":"/paper/neural-architecture-search-for-lf-mmi-trained","slug":"neural-architecture-search-for-lf-mmi-trained","title":"Neural Architecture Search For LF-MMI Trained Time Delay Neural Networks","date":"2022-01-08","arxiv_id":"2201.03943","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-datasets-in","slug":"automatic-speech-recognition-datasets-in","title":"Automatic Speech Recognition Datasets in Cantonese: A Survey and New Dataset","date":"2022-01-07","arxiv_id":"2201.02419","repositories_listed":1,"syntology":null},{"url":"/paper/improving-mandarin-end-to-end-speech","slug":"improving-mandarin-end-to-end-speech","title":"Improving Mandarin End-to-End Speech Recognition with Word N-gram Language Model","date":"2022-01-06","arxiv_id":"2201.01995","repositories_listed":1,"syntology":null},{"url":"/paper/robust-self-supervised-audio-visual-speech","slug":"robust-self-supervised-audio-visual-speech","title":"Robust Self-Supervised Audio-Visual Speech Recognition","date":"2022-01-05","arxiv_id":"2201.01763","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-self-supervised-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2201.01763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01763"}},"official":{"repos":["facebookresearch/av_hubert"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/a-hierarchical-model-for-spoken-language","slug":"a-hierarchical-model-for-spoken-language","title":"A Discriminative Hierarchical PLDA-based Model for Spoken Language Recognition","date":"2022-01-04","arxiv_id":"2201.01364","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-attacks-against-windows-pe","slug":"adversarial-attacks-against-windows-pe","title":"Adversarial Attacks against Windows PE Malware Detection: A Survey of the State-of-the-Art","date":"2021-12-23","arxiv_id":"2112.12310","repositories_listed":1,"syntology":null},{"url":"/paper/regularizing-end-to-end-speech-translation","slug":"regularizing-end-to-end-speech-translation","title":"Regularizing End-to-End Speech Translation with Triangular Decomposition Agreement","date":"2021-12-21","arxiv_id":"2112.10991","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-for-monolingual-end-to-end","slug":"continual-learning-for-monolingual-end-to-end","title":"Continual Learning for Monolingual End-to-End Automatic Speech Recognition","date":"2021-12-17","arxiv_id":"2112.09427","repositories_listed":1,"syntology":null},{"url":"/paper/automated-deep-learning-neural-architecture","slug":"automated-deep-learning-neural-architecture","title":"Automated Deep Learning: Neural Architecture Search Is Not the End","date":"2021-12-16","arxiv_id":"2112.09245","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automated-deep-learning-neural-architecture#ran","syntology_url":"https://syntology.ai/paper/2112.09245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.09245"}},"official":{"repos":["D-X-Y/Awesome-AutoDL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-learning-for-speech-1","slug":"self-supervised-learning-for-speech-1","title":"Self-Supervised Learning for speech recognition with Intermediate layer supervision","date":"2021-12-16","arxiv_id":"2112.08778","repositories_listed":1,"syntology":null},{"url":"/paper/importantaug-a-data-augmentation-agent-for","slug":"importantaug-a-data-augmentation-agent-for","title":"ImportantAug: a data augmentation agent for speech","date":"2021-12-14","arxiv_id":"2112.07156","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-use-of-external-data-for-spoken-named","slug":"on-the-use-of-external-data-for-spoken-named","title":"On the Use of External Data for Spoken Named Entity Recognition","date":"2021-12-14","arxiv_id":"2112.07648","repositories_listed":1,"syntology":null},{"url":"/paper/x-vector-based-voice-activity-detection-for-1","slug":"x-vector-based-voice-activity-detection-for-1","title":"X-Vector based voice activity detection for multi-genre broadcast speech-to-text","date":"2021-12-09","arxiv_id":"2112.05016","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-training-and-decoding-for-end-to","slug":"consistent-training-and-decoding-for-end-to","title":"Consistent Training and Decoding For End-to-end Speech Recognition Using Lattice-free MMI","date":"2021-12-05","arxiv_id":"2112.02498","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-adaptive-multiscale-temporal","slug":"understanding-adaptive-multiscale-temporal","title":"Understanding Adaptive, Multiscale Temporal Integration In Deep Speech Recognition Systems","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/romanian-speech-recognition-experiments-from","slug":"romanian-speech-recognition-experiments-from","title":"Romanian Speech Recognition Experiments from the ROBIN Project","date":"2021-11-23","arxiv_id":"2111.12028","repositories_listed":1,"syntology":null},{"url":"/paper/slue-new-benchmark-tasks-for-spoken-language","slug":"slue-new-benchmark-tasks-for-spoken-language","title":"SLUE: New Benchmark Tasks for Spoken Language Understanding Evaluation on Natural Speech","date":"2021-11-19","arxiv_id":"2111.10367","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slue-new-benchmark-tasks-for-spoken-language#ran","syntology_url":"https://syntology.ai/paper/2111.10367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10367"}},"official":{"repos":["asappresearch/slue-toolkit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/integrated-semantic-and-phonetic-post-1","slug":"integrated-semantic-and-phonetic-post-1","title":"Integrated Semantic and Phonetic Post-correction for Chinese Speech Recognition","date":"2021-11-16","arxiv_id":"2111.08400","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-semantic-driven-phoneme","slug":"self-supervised-semantic-driven-phoneme","title":"Self-supervised Semantic-driven Phoneme Discovery for Zero-resource Speech Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-contribution-of-multiple-model","slug":"measuring-the-contribution-of-multiple-model","title":"Measuring the Contribution of Multiple Model Representations in Detecting Adversarial Instances","date":"2021-11-13","arxiv_id":"2111.07035","repositories_listed":1,"syntology":null},{"url":"/paper/asl-trigger-recognition-in-mixed-activity","slug":"asl-trigger-recognition-in-mixed-activity","title":"ASL Trigger Recognition in Mixed Activity/Signing Sequences for RF Sensor-Based User Interfaces","date":"2021-11-10","arxiv_id":"2111.05480","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-randomized-smoothing-for-1","slug":"sequential-randomized-smoothing-for-1","title":"Sequential Randomized Smoothing for Adversarially Robust Speech Recognition","date":"2021-11-05","arxiv_id":"2112.03000","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-to-sequence-modeling-for-action-1","slug":"sequence-to-sequence-modeling-for-action-1","title":"Sequence-to-Sequence Modeling for Action Identification at High Temporal Resolution","date":"2021-11-03","arxiv_id":"2111.02521","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sequence-to-sequence-modeling-for-action-1#ran","syntology_url":"https://syntology.ai/paper/2111.02521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02521"}},"official":null}},{"url":"/paper/cross-lingual-transfer-for-speech-processing","slug":"cross-lingual-transfer-for-speech-processing","title":"Cross-lingual Transfer for Speech Processing using Acoustic Language Similarity","date":"2021-11-02","arxiv_id":"2111.01326","repositories_listed":1,"syntology":null},{"url":"/paper/a-transfer-learning-based-approach-for","slug":"a-transfer-learning-based-approach-for","title":"A transfer learning based approach for pronunciation scoring","date":"2021-11-01","arxiv_id":"2111.00976","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-augmented-transducer-networks","slug":"cross-attention-augmented-transducer-networks","title":"Cross Attention Augmented Transducer Networks for Simultaneous Translation","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-robustness-of-you-only-hear-once","slug":"evaluating-robustness-of-you-only-hear-once","title":"Evaluating robustness of You Only Hear Once(YOHO) Algorithm on noisy audios in the VOICe Dataset","date":"2021-11-01","arxiv_id":"2111.01205","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-evaluation-of-language-models-for","slug":"intrinsic-evaluation-of-language-models-for","title":"Intrinsic evaluation of language models for code-switching","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/orthographic-transliteration-for-kabyle","slug":"orthographic-transliteration-for-kabyle","title":"Orthographic Transliteration for Kabyle Speech Recognition","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revealing-and-protecting-labels-in","slug":"revealing-and-protecting-labels-in","title":"Revealing and Protecting Labels in Distributed Training","date":"2021-10-31","arxiv_id":"2111.00556","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-joint-decoding-based-multi-talker","slug":"revisiting-joint-decoding-based-multi-talker","title":"Revisiting joint decoding based multi-talker speech recognition with DNN acoustic model","date":"2021-10-31","arxiv_id":"2111.00009","repositories_listed":1,"syntology":null},{"url":"/paper/game-of-gradients-mitigating-irrelevant","slug":"game-of-gradients-mitigating-irrelevant","title":"Game of Gradients: Mitigating Irrelevant Clients in Federated Learning","date":"2021-10-23","arxiv_id":"2110.12257","repositories_listed":1,"syntology":null},{"url":"/paper/aequevox-automated-fairness-testing-of-speech","slug":"aequevox-automated-fairness-testing-of-speech","title":"AequeVox: Automated Fairness Testing of Speech Recognition Systems","date":"2021-10-19","arxiv_id":"2110.09843","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-of-french-phonetic-idiosyncrasies","slug":"analysis-of-french-phonetic-idiosyncrasies","title":"Analysis of French Phonetic Idiosyncrasies for Accent Recognition","date":"2021-10-18","arxiv_id":"2110.09179","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-speaker-adaptation-approach-for-asr","slug":"a-unified-speaker-adaptation-approach-for-asr","title":"A Unified Speaker Adaptation Approach for ASR","date":"2021-10-16","arxiv_id":"2110.08545","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-introductions-in-podcast-episodes","slug":"identifying-introductions-in-podcast-episodes","title":"Identifying Introductions in Podcast Episodes from Automatically Generated Transcripts","date":"2021-10-14","arxiv_id":"2110.07096","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-wav2vec-2-0-fine-tuning-for","slug":"exploring-wav2vec-2-0-fine-tuning-for","title":"Exploring Wav2vec 2.0 fine-tuning for improved speech emotion recognition","date":"2021-10-12","arxiv_id":"2110.06309","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-wav2vec-2-0-fine-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2110.06309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06309"}},"official":{"repos":["b04901014/FT-w2v2-ser"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lightseq-accelerated-training-for-transformer","slug":"lightseq-accelerated-training-for-transformer","title":"LightSeq2: Accelerated Training for Transformer-based Models on GPUs","date":"2021-10-12","arxiv_id":"2110.05722","repositories_listed":1,"syntology":null},{"url":"/paper/long-expressive-memory-for-sequence-modeling-1","slug":"long-expressive-memory-for-sequence-modeling-1","title":"Long Expressive Memory for Sequence Modeling","date":"2021-10-10","arxiv_id":"2110.04744","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":1,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/long-expressive-memory-for-sequence-modeling-1#ran","syntology_url":"https://syntology.ai/paper/2110.04744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04744"}},"official":{"repos":["tk-rusch/lem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/arabic-speech-emotion-recognition-employing","slug":"arabic-speech-emotion-recognition-employing","title":"Arabic Speech Emotion Recognition Employing Wav2vec2.0 and HuBERT Based on BAVED Dataset","date":"2021-10-09","arxiv_id":"2110.04425","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-conditional-end-to-end-asr-with","slug":"hierarchical-conditional-end-to-end-asr-with","title":"Hierarchical Conditional End-to-End ASR with CTC and Multi-Granular Subword Units","date":"2021-10-08","arxiv_id":"2110.04109","repositories_listed":1,"syntology":null},{"url":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","repositories_listed":1,"syntology":null},{"url":"/paper/late-reverberation-suppression-using-u-nets","slug":"late-reverberation-suppression-using-u-nets","title":"Late reverberation suppression using U-nets","date":"2021-10-05","arxiv_id":"2110.02144","repositories_listed":1,"syntology":null},{"url":"/paper/federated-learning-in-asr-not-as-easy-as-you","slug":"federated-learning-in-asr-not-as-easy-as-you","title":"Federated Learning in ASR: Not as Easy as You Think","date":"2021-09-30","arxiv_id":"2109.15108","repositories_listed":1,"syntology":null},{"url":"/paper/fastcorrect-2-fast-error-correction-on","slug":"fastcorrect-2-fast-error-correction-on","title":"FastCorrect 2: Fast Error Correction on Multiple Candidates for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":"2109.14420","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fastcorrect-2-fast-error-correction-on#ran","syntology_url":"https://syntology.ai/paper/2109.14420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14420"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-robust-r-u-evaluating-task-oriented","slug":"how-robust-r-u-evaluating-task-oriented","title":"\"How Robust r u?\": Evaluating Task-Oriented Dialogue Systems on Spoken Conversations","date":"2021-09-28","arxiv_id":"2109.13489","repositories_listed":1,"syntology":null},{"url":"/paper/factorized-neural-transducer-for-efficient","slug":"factorized-neural-transducer-for-efficient","title":"Factorized Neural Transducer for Efficient Language Model Adaptation","date":"2021-09-27","arxiv_id":"2110.01500","repositories_listed":1,"syntology":null},{"url":"/paper/fast-md-fast-multi-decoder-end-to-end-speech","slug":"fast-md-fast-multi-decoder-end-to-end-speech","title":"Fast-MD: Fast Multi-Decoder End-to-End Speech Translation with Non-Autoregressive Hidden Intermediates","date":"2021-09-27","arxiv_id":"2109.12804","repositories_listed":1,"syntology":null},{"url":"/paper/sd-qa-spoken-dialectal-question-answering-for","slug":"sd-qa-spoken-dialectal-question-answering-for","title":"SD-QA: Spoken Dialectal Question Answering for the Real World","date":"2021-09-24","arxiv_id":"2109.12072","repositories_listed":1,"syntology":{"n":22,"n_ran":20,"n_constructed":0,"n_ran_checked":20,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":0,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sd-qa-spoken-dialectal-question-answering-for#ran","syntology_url":"https://syntology.ai/paper/2109.12072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.12072"}},"official":{"repos":["ffaisal93/sd-qa"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":20,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robustness-analysis-of-deep-learning","slug":"robustness-analysis-of-deep-learning","title":"Robustness Analysis of Deep Learning Frameworks on Mobile Platforms","date":"2021-09-20","arxiv_id":"2109.09869","repositories_listed":1,"syntology":null},{"url":"/paper/ai-accelerator-survey-and-trends","slug":"ai-accelerator-survey-and-trends","title":"AI Accelerator Survey and Trends","date":"2021-09-18","arxiv_id":"2109.08957","repositories_listed":1,"syntology":null},{"url":"/paper/performance-efficiency-trade-offs-in","slug":"performance-efficiency-trade-offs-in","title":"Performance-Efficiency Trade-offs in Unsupervised Pre-training for Speech Recognition","date":"2021-09-14","arxiv_id":"2109.06870","repositories_listed":1,"syntology":null},{"url":"/paper/multi-sentence-resampling-a-simple-approach","slug":"multi-sentence-resampling-a-simple-approach","title":"Multi-Sentence Resampling: A Simple Approach to Alleviate Dataset Length Bias and Beam-Search Degradation","date":"2021-09-13","arxiv_id":"2109.06253","repositories_listed":1,"syntology":null},{"url":"/paper/datasets-a-community-library-for-natural","slug":"datasets-a-community-library-for-natural","title":"Datasets: A Community Library for Natural Language Processing","date":"2021-09-07","arxiv_id":"2109.02846","repositories_listed":1,"syntology":null},{"url":"/paper/complementing-handcrafted-features-with-raw","slug":"complementing-handcrafted-features-with-raw","title":"Complementing Handcrafted Features with Raw Waveform Using a Light-weight Auxiliary Model","date":"2021-09-06","arxiv_id":"2109.02773","repositories_listed":1,"syntology":null},{"url":"/paper/crypten-secure-multi-party-computation-meets","slug":"crypten-secure-multi-party-computation-meets","title":"CrypTen: Secure Multi-Party Computation Meets Machine Learning","date":"2021-09-02","arxiv_id":"2109.00984","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/crypten-secure-multi-party-computation-meets#ran","syntology_url":"https://syntology.ai/paper/2109.00984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00984"}},"official":null}},{"url":"/paper/vietnamese-end-to-end-speech-recognition","slug":"vietnamese-end-to-end-speech-recognition","title":"Vietnamese end-to-end speech recognition using wav2vec 2.0","date":"2021-09-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/you-only-hear-once-a-yolo-like-algorithm-for","slug":"you-only-hear-once-a-yolo-like-algorithm-for","title":"You Only Hear Once: A YOLO-like Algorithm for Audio Segmentation and Sound Event Detection","date":"2021-09-01","arxiv_id":"2109.00962","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-conformer-progressive-downsampling","slug":"efficient-conformer-progressive-downsampling","title":"Efficient conformer: Progressive downsampling and grouped attention for automatic speech recognition","date":"2021-08-31","arxiv_id":"2109.01163","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-with-joint","slug":"end-to-end-speech-recognition-with-joint","title":"End-to-End Speech Recognition With Joint Dereverberation Of Sub-Band Autoregressive Envelopes","date":"2021-08-09","arxiv_id":"2108.03975","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-distillation-from-bert-transformer","slug":"knowledge-distillation-from-bert-transformer","title":"Knowledge Distillation from BERT Transformer to Speech Transformer for Intent Classification","date":"2021-08-05","arxiv_id":"2108.02598","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-multilingual-end-to-end-speech","slug":"a-study-of-multilingual-end-to-end-speech","title":"A Study of Multilingual End-to-End Speech Recognition for Kazakh, Russian, and English","date":"2021-08-03","arxiv_id":"2108.01280","repositories_listed":1,"syntology":null},{"url":"/paper/the-history-of-speech-recognition-to-the-year","slug":"the-history-of-speech-recognition-to-the-year","title":"The History of Speech Recognition to the Year 2030","date":"2021-07-30","arxiv_id":"2108.00084","repositories_listed":1,"syntology":null},{"url":"/paper/usc-an-open-source-uzbek-speech-corpus-and","slug":"usc-an-open-source-uzbek-speech-corpus-and","title":"USC: An Open-Source Uzbek Speech Corpus and Initial Speech Recognition Experiments","date":"2021-07-30","arxiv_id":"2107.14419","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-allophone-graphs-for-language","slug":"differentiable-allophone-graphs-for-language","title":"Differentiable Allophone Graphs for Language-Universal Speech Recognition","date":"2021-07-24","arxiv_id":"2107.11628","repositories_listed":1,"syntology":null},{"url":"/paper/brazilian-portuguese-speech-recognition-using","slug":"brazilian-portuguese-speech-recognition-using","title":"Brazilian Portuguese Speech Recognition Using Wav2vec 2.0","date":"2021-07-23","arxiv_id":"2107.11414","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-spatial-and-spectral-1","slug":"bridging-the-gap-between-spatial-and-spectral-1","title":"Bridging the Gap between Spatial and Spectral Domains: A Unified Framework for Graph Neural Networks","date":"2021-07-21","arxiv_id":"2107.10234","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-model-with-self-adaptive-sliding","slug":"sequence-model-with-self-adaptive-sliding","title":"Sequence Model with Self-Adaptive Sliding Window for Efficient Spoken Document Segmentation","date":"2021-07-20","arxiv_id":"2107.09278","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-end-to-end-asr-based-on-blockwise","slug":"streaming-end-to-end-asr-based-on-blockwise","title":"Streaming End-to-End ASR based on Blockwise Non-Autoregressive Models","date":"2021-07-20","arxiv_id":"2107.09428","repositories_listed":1,"syntology":null},{"url":"/paper/token-level-supervised-contrastive-learning","slug":"token-level-supervised-contrastive-learning","title":"Token-Level Supervised Contrastive Learning for Punctuation Restoration","date":"2021-07-19","arxiv_id":"2107.09099","repositories_listed":1,"syntology":null},{"url":"/paper/strode-stochastic-boundary-ordinary","slug":"strode-stochastic-boundary-ordinary","title":"STRODE: Stochastic Boundary Ordinary Differential Equation","date":"2021-07-17","arxiv_id":"2107.08273","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/strode-stochastic-boundary-ordinary#ran","syntology_url":"https://syntology.ai/paper/2107.08273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08273"}},"official":{"repos":["Waffle-Liu/STRODE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multilingual-and-crosslingual-speech","slug":"multilingual-and-crosslingual-speech","title":"Multilingual and crosslingual speech recognition using phonological-vector based phone embeddings","date":"2021-07-11","arxiv_id":"2107.05038","repositories_listed":1,"syntology":null},{"url":"/paper/layer-wise-analysis-of-a-self-supervised","slug":"layer-wise-analysis-of-a-self-supervised","title":"Layer-wise Analysis of a Self-supervised Speech Representation Model","date":"2021-07-10","arxiv_id":"2107.04734","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-ctc-crf-based-end-to-end-speech","slug":"advancing-ctc-crf-based-end-to-end-speech","title":"Advancing CTC-CRF Based End-to-End Speech Recognition with Wordpieces and Conformers","date":"2021-07-07","arxiv_id":"2107.03007","repositories_listed":1,"syntology":null},{"url":"/paper/kosp2e-korean-speech-to-english-translation","slug":"kosp2e-korean-speech-to-english-translation","title":"Kosp2e: Korean Speech to English Translation Corpus","date":"2021-07-06","arxiv_id":"2107.02875","repositories_listed":1,"syntology":null},{"url":"/paper/instant-one-shot-word-learning-for-context","slug":"instant-one-shot-word-learning-for-context","title":"Instant One-Shot Word-Learning for Context-Specific Neural Sequence-to-Sequence Speech Recognition","date":"2021-07-05","arxiv_id":"2107.02268","repositories_listed":1,"syntology":null},{"url":"/paper/tenet-a-time-reversal-enhancement-network-for","slug":"tenet-a-time-reversal-enhancement-network-for","title":"TENET: A Time-reversal Enhancement Network for Noise-robust ASR","date":"2021-07-04","arxiv_id":"2107.01531","repositories_listed":1,"syntology":null},{"url":"/paper/relaxed-attention-a-simple-method-to-boost","slug":"relaxed-attention-a-simple-method-to-boost","title":"Relaxed Attention: A Simple Method to Boost Performance of End-to-End Automatic Speech Recognition","date":"2021-07-02","arxiv_id":"2107.01275","repositories_listed":1,"syntology":null},{"url":"/paper/combining-frame-synchronous-and-label","slug":"combining-frame-synchronous-and-label","title":"Combining Frame-Synchronous and Label-Synchronous Systems for Speech Recognition","date":"2021-07-01","arxiv_id":"2107.00764","repositories_listed":1,"syntology":null},{"url":"/paper/pretext-tasks-selection-for-multitask-self","slug":"pretext-tasks-selection-for-multitask-self","title":"Pretext Tasks selection for multitask self-supervised speech representation learning","date":"2021-07-01","arxiv_id":"2107.00594","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pretext-tasks-selection-for-multitask-self#ran","syntology_url":"https://syntology.ai/paper/2107.00594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00594"}},"official":{"repos":["salah-zaiem/PL-groupselection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-deep-learning-a-survey-on-making","slug":"efficient-deep-learning-a-survey-on-making","title":"Efficient Deep Learning: A Survey on Making Deep Learning Models Smaller, Faster, and Better","date":"2021-06-16","arxiv_id":"2106.08962","repositories_listed":1,"syntology":null},{"url":"/paper/momentum-pseudo-labeling-for-semi-supervised","slug":"momentum-pseudo-labeling-for-semi-supervised","title":"Momentum Pseudo-Labeling for Semi-Supervised Speech Recognition","date":"2021-06-16","arxiv_id":"2106.08922","repositories_listed":1,"syntology":null},{"url":"/paper/multi-speaker-asr-combining-non","slug":"multi-speaker-asr-combining-non","title":"Multi-Speaker ASR Combining Non-Autoregressive Conformer CTC and Conditional Speaker Chain","date":"2021-06-16","arxiv_id":"2106.08595","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-sentence-similarity-size-does-not","slug":"semantic-sentence-similarity-size-does-not","title":"Semantic sentence similarity: size does not always matter","date":"2021-06-16","arxiv_id":"2106.08648","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-use-of-prosody-in-constituency","slug":"assessing-the-use-of-prosody-in-constituency","title":"Assessing the Use of Prosody in Constituency Parsing of Imperfect Transcripts","date":"2021-06-14","arxiv_id":"2106.07794","repositories_listed":1,"syntology":null},{"url":"/paper/learning-audio-visual-dereverberation","slug":"learning-audio-visual-dereverberation","title":"Learning Audio-Visual Dereverberation","date":"2021-06-14","arxiv_id":"2106.07732","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-external-pos-tagger-for","slug":"incorporating-external-pos-tagger-for","title":"Incorporating External POS Tagger for Punctuation Restoration","date":"2021-06-12","arxiv_id":"2106.06731","repositories_listed":1,"syntology":null},{"url":"/paper/muddling-label-regularization-deep-learning","slug":"muddling-label-regularization-deep-learning","title":"Muddling Label Regularization: Deep Learning for Tabular Datasets","date":"2021-06-08","arxiv_id":"2106.04462","repositories_listed":1,"syntology":null},{"url":"/paper/cape-encoding-relative-positions-with","slug":"cape-encoding-relative-positions-with","title":"CAPE: Encoding Relative Positions with Continuous Augmented Positional Embeddings","date":"2021-06-06","arxiv_id":"2106.03143","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-contextual-language-model","slug":"attention-based-contextual-language-model","title":"Attention-based Contextual Language Model Adaptation for Speech Recognition","date":"2021-06-02","arxiv_id":"2106.01451","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/attention-based-contextual-language-model#ran","syntology_url":"https://syntology.ai/paper/2106.01451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01451"}},"official":{"repos":["amazon-research/contextual-attention-nlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-speech-recognition-in-sanskrit-a","slug":"automatic-speech-recognition-in-sanskrit-a","title":"Automatic Speech Recognition in Sanskrit: A New Speech Corpus and Modelling Insights","date":"2021-06-02","arxiv_id":"2106.05852","repositories_listed":1,"syntology":null},{"url":"/paper/byakto-speech-real-time-long-speech-synthesis","slug":"byakto-speech-real-time-long-speech-synthesis","title":"Byakto Speech: Real-time long speech synthesis with convolutional neural network: Transfer learning from English to Bangla","date":"2021-05-31","arxiv_id":"2106.03937","repositories_listed":1,"syntology":null},{"url":"/paper/quantization-and-deployment-of-deep-neural","slug":"quantization-and-deployment-of-deep-neural","title":"Quantization and Deployment of Deep Neural Networks on Microcontrollers","date":"2021-05-27","arxiv_id":"2105.13331","repositories_listed":1,"syntology":null},{"url":"/paper/attack-on-practical-speaker-verification","slug":"attack-on-practical-speaker-verification","title":"Attack on practical speaker verification system using universal adversarial perturbations","date":"2021-05-19","arxiv_id":"2105.09022","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/attack-on-practical-speaker-verification#ran","syntology_url":"https://syntology.ai/paper/2105.09022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.09022"}},"official":{"repos":["zhang-wy15/Attack_practical_asv"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/hardware-synthesis-of-state-space-equations","slug":"hardware-synthesis-of-state-space-equations","title":"Hardware Synthesis of State-Space Equations; Application to FPGA Implementation of Shallow and Deep Neural Networks","date":"2021-05-15","arxiv_id":"2105.07131","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-reordering-capability-in","slug":"investigating-the-reordering-capability-in","title":"Investigating the Reordering Capability in CTC-based Non-Autoregressive End-to-End Speech Translation","date":"2021-05-11","arxiv_id":"2105.04840","repositories_listed":1,"syntology":null},{"url":"/paper/fastcorrect-fast-error-correction-with-edit","slug":"fastcorrect-fast-error-correction-with-edit","title":"FastCorrect: Fast Error Correction with Edit Alignment for Automatic Speech Recognition","date":"2021-05-09","arxiv_id":"2105.03842","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/fastcorrect-fast-error-correction-with-edit#ran","syntology_url":"https://syntology.ai/paper/2105.03842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03842"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/speechmoe-scaling-to-large-acoustic-models","slug":"speechmoe-scaling-to-large-acoustic-models","title":"SpeechMoE: Scaling to Large Acoustic Models with Dynamic Routing Mixture of Experts","date":"2021-05-07","arxiv_id":"2105.03036","repositories_listed":1,"syntology":null},{"url":"/paper/software-engineering-for-ai-based-systems-a","slug":"software-engineering-for-ai-based-systems-a","title":"Software Engineering for AI-Based Systems: A Survey","date":"2021-05-05","arxiv_id":"2105.01984","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-from-federated","slug":"end-to-end-speech-recognition-from-federated","title":"End-to-End Speech Recognition from Federated Acoustic Models","date":"2021-04-29","arxiv_id":"2104.14297","repositories_listed":1,"syntology":null}],"record_sha256":"4140cbe5d7c1f82b4421441710dbb2083f5dd26b181cc5a9bb7b79ed828cd626","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}