{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/9","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":65,"rows_per_page":100,"rows":[801,900],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/8","next":"/task/speech-recognition/papers/10","papers":[{"url":"/paper/dent-ddsp-data-efficient-noisy-speech","slug":"dent-ddsp-data-efficient-noisy-speech","title":"DENT-DDSP: Data-efficient noisy speech generator using differentiable digital signal processors for explicit distortion modelling and noise-robust speech recognition","date":"2022-08-01","arxiv_id":"2208.00987","repositories_listed":1,"syntology":null},{"url":"/paper/domain-specific-wav2vec-2-0-fine-tuning-for","slug":"domain-specific-wav2vec-2-0-fine-tuning-for","title":"Domain Specific Wav2vec 2.0 Fine-tuning For The SE&R 2022 Challenge","date":"2022-07-29","arxiv_id":"2207.14418","repositories_listed":1,"syntology":null},{"url":"/paper/soundchoice-grapheme-to-phoneme-models-with","slug":"soundchoice-grapheme-to-phoneme-models-with","title":"SoundChoice: Grapheme-to-Phoneme Models with Semantic Disambiguation","date":"2022-07-27","arxiv_id":"2207.13703","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-recurrent-units-and-the-forward","slug":"bayesian-recurrent-units-and-the-forward","title":"Bayesian Recurrent Units and the Forward-Backward Algorithm","date":"2022-07-21","arxiv_id":"2207.10486","repositories_listed":1,"syntology":null},{"url":"/paper/autodice-fully-automated-distributed-cnn","slug":"autodice-fully-automated-distributed-cnn","title":"AutoDiCE: Fully Automated Distributed CNN Inference at the Edge","date":"2022-07-20","arxiv_id":"2207.12113","repositories_listed":1,"syntology":null},{"url":"/paper/towards-transfer-learning-of-wav2vec-2-0-for","slug":"towards-transfer-learning-of-wav2vec-2-0-for","title":"Transfer Learning of wav2vec 2.0 for Automatic Lyric Transcription","date":"2022-07-20","arxiv_id":"2207.09747","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-transfer-learning-of-wav2vec-2-0-for#ran","syntology_url":"https://syntology.ai/paper/2207.09747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09747"}},"official":{"repos":["guxm2021/alt_speechbrain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/when-is-tts-augmentation-through-a-pivot","slug":"when-is-tts-augmentation-through-a-pivot","title":"When Is TTS Augmentation Through a Pivot Language Useful?","date":"2022-07-20","arxiv_id":"2207.09889","repositories_listed":1,"syntology":null},{"url":"/paper/espnet-se-speech-enhancement-for-robust","slug":"espnet-se-speech-enhancement-for-robust","title":"ESPnet-SE++: Speech Enhancement for Robust Speech Recognition, Translation, and Understanding","date":"2022-07-19","arxiv_id":"2207.09514","repositories_listed":1,"syntology":null},{"url":"/paper/position-prediction-as-an-effective","slug":"position-prediction-as-an-effective","title":"Position Prediction as an Effective Pretraining Strategy","date":"2022-07-15","arxiv_id":"2207.07611","repositories_listed":1,"syntology":null},{"url":"/paper/a-single-self-supervised-model-for-many","slug":"a-single-self-supervised-model-for-many","title":"u-HuBERT: Unified Mixed-Modal Speech Pretraining And Zero-Shot Transfer to Unlabeled Modality","date":"2022-07-14","arxiv_id":"2207.07036","repositories_listed":1,"syntology":null},{"url":"/paper/mm-alt-a-multimodal-automatic-lyric","slug":"mm-alt-a-multimodal-automatic-lyric","title":"MM-ALT: A Multimodal Automatic Lyric Transcription System","date":"2022-07-13","arxiv_id":"2207.06127","repositories_listed":1,"syntology":null},{"url":"/paper/visual-context-driven-audio-feature","slug":"visual-context-driven-audio-feature","title":"Visual Context-driven Audio Feature Enhancement for Robust End-to-End Audio-Visual Speech Recognition","date":"2022-07-13","arxiv_id":"2207.06020","repositories_listed":1,"syntology":null},{"url":"/paper/speaker-anonymization-with-phonetic","slug":"speaker-anonymization-with-phonetic","title":"Speaker Anonymization with Phonetic Intermediate Representations","date":"2022-07-11","arxiv_id":"2207.04834","repositories_listed":1,"syntology":null},{"url":"/paper/cprune-compiler-informed-model-pruning-for","slug":"cprune-compiler-informed-model-pruning-for","title":"CPrune: Compiler-Informed Model Pruning for Efficient Target-Aware DNN Execution","date":"2022-07-04","arxiv_id":"2207.01260","repositories_listed":1,"syntology":null},{"url":"/paper/generating-gender-ambiguous-voices-for","slug":"generating-gender-ambiguous-voices-for","title":"Generating gender-ambiguous voices for privacy-preserving speech recognition","date":"2022-07-03","arxiv_id":"2207.01052","repositories_listed":1,"syntology":null},{"url":"/paper/behancepr-a-punctuation-restoration-dataset","slug":"behancepr-a-punctuation-restoration-dataset","title":"BehancePR: A Punctuation Restoration Dataset for Livestreaming Video Transcript","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/nextformer-a-convnext-augmented-conformer-for","slug":"nextformer-a-convnext-augmented-conformer-for","title":"Nextformer: A ConvNeXt Augmented Conformer For End-To-End Speech Recognition","date":"2022-06-29","arxiv_id":"2206.14747","repositories_listed":1,"syntology":null},{"url":"/paper/stop-a-dataset-for-spoken-task-oriented","slug":"stop-a-dataset-for-spoken-task-oriented","title":"STOP: A dataset for Spoken Task Oriented Semantic Parsing","date":"2022-06-29","arxiv_id":"2207.10643","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-a-pretrained-language-model-to-a","slug":"distilling-a-pretrained-language-model-to-a","title":"Distilling a Pretrained Language Model to a Multilingual ASR Model","date":"2022-06-25","arxiv_id":"2206.12638","repositories_listed":1,"syntology":null},{"url":"/paper/a-systematic-comparison-of-phonetic-aware","slug":"a-systematic-comparison-of-phonetic-aware","title":"A Systematic Comparison of Phonetic Aware Techniques for Speech Enhancement","date":"2022-06-22","arxiv_id":"2206.11000","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-cross-domain-speech-recognition-with","slug":"boosting-cross-domain-speech-recognition-with","title":"Boosting Cross-Domain Speech Recognition with Self-Supervision","date":"2022-06-20","arxiv_id":"2206.09783","repositories_listed":1,"syntology":null},{"url":"/paper/avatar-unconstrained-audiovisual-speech","slug":"avatar-unconstrained-audiovisual-speech","title":"AVATAR: Unconstrained Audiovisual Speech Recognition","date":"2022-06-15","arxiv_id":"2206.07684","repositories_listed":1,"syntology":null},{"url":"/paper/nntrainer-light-weight-on-device-training","slug":"nntrainer-light-weight-on-device-training","title":"A New Frontier of AI: On-Device AI Training and Personalization","date":"2022-06-09","arxiv_id":"2206.04688","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-end-to-end-speech-to-text","slug":"revisiting-end-to-end-speech-to-text","title":"Revisiting End-to-End Speech-to-Text Translation From Scratch","date":"2022-06-09","arxiv_id":"2206.04571","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/revisiting-end-to-end-speech-to-text#ran","syntology_url":"https://syntology.ai/paper/2206.04571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04571"}},"official":{"repos":["bzhangGo/zero"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/lae-language-aware-encoder-for-monolingual","slug":"lae-language-aware-encoder-for-monolingual","title":"LAE: Language-Aware Encoder for Monolingual and Multilingual ASR","date":"2022-06-05","arxiv_id":"2206.02093","repositories_listed":1,"syntology":null},{"url":"/paper/variable-rate-hierarchical-cpc-leads-to","slug":"variable-rate-hierarchical-cpc-leads-to","title":"Variable-rate hierarchical CPC leads to acoustic unit discovery in speech","date":"2022-06-05","arxiv_id":"2206.02211","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/variable-rate-hierarchical-cpc-leads-to#ran","syntology_url":"https://syntology.ai/paper/2206.02211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02211"}},"official":{"repos":["chorowski-lab/hcpc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ci-avsr-a-cantonese-audio-visual-speech-1","slug":"ci-avsr-a-cantonese-audio-visual-speech-1","title":"CI-AVSR: A Cantonese Audio-Visual Speech Datasetfor In-car Command Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ssr7000-a-synchronized-corpus-of-ultrasound","slug":"ssr7000-a-synchronized-corpus-of-ultrasound","title":"SSR7000: A Synchronized Corpus of Ultrasound Tongue Imaging for End-to-End Silent Speech Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/global-normalization-for-streaming-speech","slug":"global-normalization-for-streaming-speech","title":"Global Normalization for Streaming Speech Recognition in a Modular Framework","date":"2022-05-26","arxiv_id":"2205.13674","repositories_listed":1,"syntology":null},{"url":"/paper/fleurs-few-shot-learning-evaluation-of","slug":"fleurs-few-shot-learning-evaluation-of","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","date":"2022-05-25","arxiv_id":"2205.12446","repositories_listed":1,"syntology":null},{"url":"/paper/training-efficient-cnns-tweaking-the-nuts-and","slug":"training-efficient-cnns-tweaking-the-nuts-and","title":"Training Efficient CNNS: Tweaking the Nuts and Bolts of Neural Networks for Lighter, Faster and Robust Models","date":"2022-05-23","arxiv_id":"2205.12050","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-with-image-descriptors-are","slug":"language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","arxiv_id":"2205.10747","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-with-image-descriptors-are#ran","syntology_url":"https://syntology.ai/paper/2205.10747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10747"}},"official":{"repos":["mikewangwzhl/vidil"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/who-are-we-talking-about-handling-person-2","slug":"who-are-we-talking-about-handling-person-2","title":"Who Are We Talking About? Handling Person Names in Speech Translation","date":"2022-05-13","arxiv_id":"2205.06755","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-enabled-semantic-communications","slug":"deep-learning-enabled-semantic-communications","title":"Deep Learning Enabled Semantic Communications with Speech Recognition and Synthesis","date":"2022-05-09","arxiv_id":"2205.04603","repositories_listed":1,"syntology":null},{"url":"/paper/vietnamese-automatic-speech-recognition-using","slug":"vietnamese-automatic-speech-recognition-using","title":"Vietnamese Automatic Speech Recognition using Wav2vec 2.0","date":"2022-05-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/wav2vec2-base-vietnamese-160h","slug":"wav2vec2-base-vietnamese-160h","title":"Wav2vec2 Base Vietnamese 160h","date":"2022-05-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-multi-aspect-multi","slug":"transformer-based-multi-aspect-multi","title":"Transformer-Based Multi-Aspect Multi-Granularity Non-Native English Speaker Pronunciation Assessment","date":"2022-05-06","arxiv_id":"2205.03432","repositories_listed":1,"syntology":null},{"url":"/paper/speaker-recognition-in-the-wild","slug":"speaker-recognition-in-the-wild","title":"Speaker Recognition in the Wild","date":"2022-05-05","arxiv_id":"2205.02475","repositories_listed":1,"syntology":null},{"url":"/paper/wav2seq-pre-training-speech-to-text-encoder","slug":"wav2seq-pre-training-speech-to-text-encoder","title":"Wav2Seq: Pre-training Speech-to-Text Encoder-Decoder Models Using Pseudo Languages","date":"2022-05-02","arxiv_id":"2205.01086","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-and-query-by","slug":"automatic-speech-recognition-and-query-by","title":"Automatic Speech Recognition and Query By Example for Creole Languages Documentation","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/how-you-say-it-matters-measuring-the-impact","slug":"how-you-say-it-matters-measuring-the-impact","title":"How You Say It Matters: Measuring the Impact of Verbal Disfluency Tags on Automated Dementia Detection","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-non-autoregressive-generation-for","slug":"a-survey-on-non-autoregressive-generation-for","title":"A Survey on Non-Autoregressive Generation for Neural Machine Translation and Beyond","date":"2022-04-20","arxiv_id":"2204.09269","repositories_listed":1,"syntology":null},{"url":"/paper/deep-transfer-learning-for-partial","slug":"deep-transfer-learning-for-partial","title":"Deep transfer operator learning for partial differential equations under conditional shift","date":"2022-04-20","arxiv_id":"2204.09810","repositories_listed":1,"syntology":null},{"url":"/paper/hubert-ee-early-exiting-hubert-for-efficient","slug":"hubert-ee-early-exiting-hubert-for-efficient","title":"HuBERT-EE: Early Exiting HuBERT for Efficient Speech Recognition","date":"2022-04-13","arxiv_id":"2204.06328","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-streaming-end-to-end-speech","slug":"large-scale-streaming-end-to-end-speech","title":"Large-Scale Streaming End-to-End Speech Translation with Neural Transducers","date":"2022-04-11","arxiv_id":"2204.05352","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-scale-streaming-end-to-end-speech#ran","syntology_url":"https://syntology.ai/paper/2204.05352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05352"}},"official":null}},{"url":"/paper/exploiting-hidden-representations-from-a-dnn","slug":"exploiting-hidden-representations-from-a-dnn","title":"Exploiting Hidden Representations from a DNN-based Speech Recogniser for Speech Intelligibility Prediction in Hearing-impaired Listeners","date":"2022-04-08","arxiv_id":"2204.04287","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-softmax-for-end-to-end-low","slug":"hierarchical-softmax-for-end-to-end-low","title":"Hierarchical Softmax for End-to-End Low-resource Multilingual Speech Recognition","date":"2022-04-08","arxiv_id":"2204.03855","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-uncertainty-measures-of","slug":"unsupervised-uncertainty-measures-of","title":"Unsupervised Uncertainty Measures of Automatic Speech Recognition for Non-intrusive Speech Intelligibility Prediction","date":"2022-04-08","arxiv_id":"2204.04288","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-uncertainty-measures-of#ran","syntology_url":"https://syntology.ai/paper/2204.04288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04288"}},"official":{"repos":["claritychallenge/clarity"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/3m-multi-loss-multi-path-and-multi-level","slug":"3m-multi-loss-multi-path-and-multi-level","title":"3M: Multi-loss, Multi-path and Multi-level Neural Networks for speech recognition","date":"2022-04-07","arxiv_id":"2204.03178","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3m-multi-loss-multi-path-and-multi-level#ran","syntology_url":"https://syntology.ai/paper/2204.03178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.03178"}},"official":{"repos":["tencent-ailab/3m-asr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emotional-speech-recognition-with-pre-trained","slug":"emotional-speech-recognition-with-pre-trained","title":"Emotional Speech Recognition with Pre-trained Deep Visual Models","date":"2022-04-06","arxiv_id":"2204.03561","repositories_listed":1,"syntology":null},{"url":"/paper/combining-spectral-and-self-supervised","slug":"combining-spectral-and-self-supervised","title":"Combining Spectral and Self-Supervised Features for Low Resource Speech Recognition and Translation","date":"2022-04-05","arxiv_id":"2204.02470","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-speech-separation-for","slug":"leveraging-speech-separation-for","title":"Low-Latency Speech Separation Guided Diarization for Telephone Conversations","date":"2022-04-05","arxiv_id":"2204.02306","repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-unsupervised-speech","slug":"towards-end-to-end-unsupervised-speech","title":"Towards End-to-end Unsupervised Speech Recognition","date":"2022-04-05","arxiv_id":"2204.02492","repositories_listed":1,"syntology":null},{"url":"/paper/primock57-a-dataset-of-primary-care-mock","slug":"primock57-a-dataset-of-primary-care-mock","title":"PriMock57: A Dataset Of Primary Care Mock Consultations","date":"2022-04-01","arxiv_id":"2204.00333","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-continuity-loss-to-reduce-over","slug":"a-hybrid-continuity-loss-to-reduce-over","title":"A Hybrid Continuity Loss to Reduce Over-Suppression for Time-domain Target Speaker Extraction","date":"2022-03-31","arxiv_id":"2203.16843","repositories_listed":1,"syntology":null},{"url":"/paper/cuside-chunking-simulating-future-context-and","slug":"cuside-chunking-simulating-future-context-and","title":"CUSIDE: Chunking, Simulating Future Context and Decoding for Streaming ASR","date":"2022-03-31","arxiv_id":"2203.16758","repositories_listed":1,"syntology":null},{"url":"/paper/hifi-vc-high-quality-asr-based-voice","slug":"hifi-vc-high-quality-asr-based-voice","title":"HiFi-VC: High Quality ASR-Based Voice Conversion","date":"2022-03-31","arxiv_id":"2203.16937","repositories_listed":1,"syntology":null},{"url":"/paper/indic-punct-an-automatic-punctuation","slug":"indic-punct-an-automatic-punctuation","title":"indic-punct: An automatic punctuation restoration and inverse text normalization framework for Indic languages","date":"2022-03-31","arxiv_id":"2203.16825","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-transformer-decoder-for-end-to","slug":"pre-training-transformer-decoder-for-end-to","title":"Pre-Training Transformer Decoder for End-to-End ASR Model with Unpaired Speech Data","date":"2022-03-31","arxiv_id":"2203.17113","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-speaker-attributed-asr-with-token","slug":"streaming-speaker-attributed-asr-with-token","title":"Streaming Speaker-Attributed ASR with Token-Level Speaker Embeddings","date":"2022-03-30","arxiv_id":"2203.16685","repositories_listed":1,"syntology":null},{"url":"/paper/using-adapters-to-overcome-catastrophic","slug":"using-adapters-to-overcome-catastrophic","title":"Using Adapters to Overcome Catastrophic Forgetting in End-to-End Automatic Speech Recognition","date":"2022-03-30","arxiv_id":"2203.16082","repositories_listed":1,"syntology":null},{"url":"/paper/vakyansh-asr-toolkit-for-low-resource-indic","slug":"vakyansh-asr-toolkit-for-low-resource-indic","title":"Vakyansh: ASR Toolkit for Low Resource Indic languages","date":"2022-03-30","arxiv_id":"2203.16512","repositories_listed":1,"syntology":null},{"url":"/paper/4-bit-conformer-with-native-quantization","slug":"4-bit-conformer-with-native-quantization","title":"4-bit Conformer with Native Quantization Aware Training for Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15952","repositories_listed":1,"syntology":null},{"url":"/paper/a-single-speaker-is-almost-all-you-need-for","slug":"a-single-speaker-is-almost-all-you-need-for","title":"ASR data augmentation in low-resource settings using cross-lingual multi-speaker TTS and cross-lingual voice conversion","date":"2022-03-29","arxiv_id":"2204.00618","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-of-eeg-frequency-bands-for","slug":"analysis-of-eeg-frequency-bands-for","title":"Analysis of EEG frequency bands for Envisioned Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15250","repositories_listed":1,"syntology":null},{"url":"/paper/earnings-22-a-practical-benchmark-for-accents","slug":"earnings-22-a-practical-benchmark-for-accents","title":"Earnings-22: A Practical Benchmark for Accents in the Wild","date":"2022-03-29","arxiv_id":"2203.15591","repositories_listed":1,"syntology":null},{"url":"/paper/integrate-lattice-free-mmi-into-end-to-end","slug":"integrate-lattice-free-mmi-into-end-to-end","title":"Integrating Lattice-Free MMI into End-to-End Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15614","repositories_listed":1,"syntology":null},{"url":"/paper/lighthubert-lightweight-and-configurable","slug":"lighthubert-lightweight-and-configurable","title":"LightHuBERT: Lightweight and Configurable Speech Representation Learning with Once-for-All Hidden-Unit BERT","date":"2022-03-29","arxiv_id":"2203.15610","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lighthubert-lightweight-and-configurable#ran","syntology_url":"https://syntology.ai/paper/2203.15610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15610"}},"official":{"repos":["mechanicalsea/lighthubert"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/shifted-chunk-encoder-for-transformer-based","slug":"shifted-chunk-encoder-for-transformer-based","title":"Shifted Chunk Encoder for Transformer Based Streaming End-to-End ASR","date":"2022-03-29","arxiv_id":"2203.15206","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-text-to-speech-synthesis-by","slug":"unsupervised-text-to-speech-synthesis-by","title":"Unsupervised Text-to-Speech Synthesis by Unsupervised Automatic Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15796","repositories_listed":1,"syntology":null},{"url":"/paper/cmgan-conformer-based-metric-gan-for-speech","slug":"cmgan-conformer-based-metric-gan-for-speech","title":"CMGAN: Conformer-based Metric GAN for Speech Enhancement","date":"2022-03-28","arxiv_id":"2203.15149","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cmgan-conformer-based-metric-gan-for-speech#ran","syntology_url":"https://syntology.ai/paper/2203.15149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15149"}},"official":{"repos":["ruizhecao96/cmgan"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-path-style-learning-for-end-to-end-noise","slug":"dual-path-style-learning-for-end-to-end-noise","title":"Dual-Path Style Learning for End-to-End Noise-Robust Speech Recognition","date":"2022-03-28","arxiv_id":"2203.14838","repositories_listed":1,"syntology":null},{"url":"/paper/finnish-parliament-asr-corpus-analysis","slug":"finnish-parliament-asr-corpus-analysis","title":"Finnish Parliament ASR corpus - Analysis, benchmarks and statistics","date":"2022-03-28","arxiv_id":"2203.14876","repositories_listed":1,"syntology":null},{"url":"/paper/a-dataset-for-speech-emotion-recognition-in","slug":"a-dataset-for-speech-emotion-recognition-in","title":"A Dataset for Speech Emotion Recognition in Greek Theatrical Plays","date":"2022-03-27","arxiv_id":"2203.15568","repositories_listed":1,"syntology":null},{"url":"/paper/towards-privacy-preserving-speech","slug":"towards-privacy-preserving-speech","title":"A Speech Representation Anonymization Framework via Selective Noise Perturbation","date":"2022-03-26","arxiv_id":"2203.14171","repositories_listed":1,"syntology":null},{"url":"/paper/flute-a-scalable-extensible-framework-for","slug":"flute-a-scalable-extensible-framework-for","title":"FLUTE: A Scalable, Extensible Framework for High-Performance Federated Learning Simulations","date":"2022-03-25","arxiv_id":"2203.13789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flute-a-scalable-extensible-framework-for#ran","syntology_url":"https://syntology.ai/paper/2203.13789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13789"}},"official":{"repos":["microsoft/msrflute"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-enhanced-and-noise-aware-networks-for","slug":"speech-enhanced-and-noise-aware-networks-for","title":"Speech-enhanced and Noise-aware Networks for Robust Speech Recognition","date":"2022-03-25","arxiv_id":"2203.13696","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-for-speech","slug":"automatic-speech-recognition-for-speech","title":"Automatic Speech Recognition for Speech Assessment of Persian Preschool Children","date":"2022-03-24","arxiv_id":"2203.12886","repositories_listed":1,"syntology":null},{"url":"/paper/neural-predictor-for-black-box-adversarial","slug":"neural-predictor-for-black-box-adversarial","title":"Neural Predictor for Black-Box Adversarial Attacks on Speech Recognition","date":"2022-03-18","arxiv_id":"2203.09849","repositories_listed":1,"syntology":null},{"url":"/paper/modelling-word-learning-and-recognition-using","slug":"modelling-word-learning-and-recognition-using","title":"Modelling word learning and recognition using visually grounded speech","date":"2022-03-14","arxiv_id":"2203.06937","repositories_listed":1,"syntology":null},{"url":"/paper/red-ace-robust-error-detection-for-asr-using-1","slug":"red-ace-robust-error-detection-for-asr-using-1","title":"RED-ACE: Robust Error Detection for ASR using Confidence Embeddings","date":"2022-03-14","arxiv_id":"2203.07172","repositories_listed":1,"syntology":null},{"url":"/paper/dual-textless-spoken-question-answering-with-1","slug":"dual-textless-spoken-question-answering-with-1","title":"DUAL: Discrete Spoken Unit Adaptive Learning for Textless Spoken Question Answering","date":"2022-03-09","arxiv_id":"2203.04911","repositories_listed":1,"syntology":null},{"url":"/paper/creating-speech-to-speech-corpus-from-dubbed","slug":"creating-speech-to-speech-corpus-from-dubbed","title":"Creating Speech-to-Speech Corpus from Dubbed Series","date":"2022-03-07","arxiv_id":"2203.03601","repositories_listed":1,"syntology":null},{"url":"/paper/towards-contextual-spelling-correction-for","slug":"towards-contextual-spelling-correction-for","title":"Towards Contextual Spelling Correction for Customization of End-to-end Speech Recognition Systems","date":"2022-03-02","arxiv_id":"2203.00888","repositories_listed":1,"syntology":null},{"url":"/paper/sentiment-word-aware-multimodal-refinement","slug":"sentiment-word-aware-multimodal-refinement","title":"Sentiment Word Aware Multimodal Refinement for Multimodal Sentiment Analysis with ASR Errors","date":"2022-03-01","arxiv_id":"2203.00257","repositories_listed":1,"syntology":null},{"url":"/paper/icassp-2022-acoustic-echo-cancellation","slug":"icassp-2022-acoustic-echo-cancellation","title":"ICASSP 2022 Acoustic Echo Cancellation Challenge","date":"2022-02-27","arxiv_id":"2202.13290","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-uni-modal-self-supervised-learning-1","slug":"leveraging-uni-modal-self-supervised-learning-1","title":"Leveraging Unimodal Self-Supervised Learning for Multimodal Audio-Visual Speech Recognition","date":"2022-02-24","arxiv_id":"2203.07996","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/leveraging-uni-modal-self-supervised-learning-1#ran","syntology_url":"https://syntology.ai/paper/2203.07996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07996"}},"official":{"repos":["lumia-group/leveraging-self-supervised-learning-for-avsr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flowsense-monitoring-airflow-in-building","slug":"flowsense-monitoring-airflow-in-building","title":"FlowSense: Monitoring Airflow in Building Ventilation Systems Using Audio Sensing","date":"2022-02-22","arxiv_id":"2202.11136","repositories_listed":1,"syntology":null},{"url":"/paper/improving-ctc-based-speech-recognition-via","slug":"improving-ctc-based-speech-recognition-via","title":"Improving CTC-based speech recognition via knowledge transferring from pre-trained language models","date":"2022-02-22","arxiv_id":"2203.03582","repositories_listed":1,"syntology":null},{"url":"/paper/spanish-and-english-phoneme-recognition-by","slug":"spanish-and-english-phoneme-recognition-by","title":"Spanish and English Phoneme Recognition by Training on Simulated Classroom Audio Recordings of Collaborative Learning Environments","date":"2022-02-21","arxiv_id":"2202.10536","repositories_listed":1,"syntology":null},{"url":"/paper/aishell-ner-named-entity-recognition-from","slug":"aishell-ner-named-entity-recognition-from","title":"AISHELL-NER: Named Entity Recognition from Chinese Speech","date":"2022-02-17","arxiv_id":"2202.08533","repositories_listed":1,"syntology":null},{"url":"/paper/adima-abuse-detection-in-multilingual-audio","slug":"adima-abuse-detection-in-multilingual-audio","title":"ADIMA: Abuse Detection In Multilingual Audio","date":"2022-02-16","arxiv_id":"2202.07991","repositories_listed":1,"syntology":null},{"url":"/paper/improving-automatic-speech-recognition-for","slug":"improving-automatic-speech-recognition-for","title":"Improving Automatic Speech Recognition for Non-Native English with Transfer Learning and Language Model Decoding","date":"2022-02-10","arxiv_id":"2202.05209","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-adapter-transfer-of-self-supervised","slug":"efficient-adapter-transfer-of-self-supervised","title":"Efficient Adapter Transfer of Self-Supervised Speech Models for Automatic Speech Recognition","date":"2022-02-07","arxiv_id":"2202.03218","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-adapter-transfer-of-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2202.03218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03218"}},"official":null}},{"url":"/paper/streaming-multi-talker-asr-with-token-level","slug":"streaming-multi-talker-asr-with-token-level","title":"Streaming Multi-Talker ASR with Token-Level Serialized Output Training","date":"2022-02-02","arxiv_id":"2202.00842","repositories_listed":1,"syntology":null},{"url":"/paper/nas-bench-suite-nas-evaluation-is-now-1","slug":"nas-bench-suite-nas-evaluation-is-now-1","title":"NAS-Bench-Suite: NAS Evaluation is (Now) Surprisingly Easy","date":"2022-01-31","arxiv_id":"2201.13396","repositories_listed":1,"syntology":null},{"url":"/paper/improving-end-to-end-contextual-speech","slug":"improving-end-to-end-contextual-speech","title":"Improving End-to-End Contextual Speech Recognition with Fine-Grained Contextual Knowledge Selection","date":"2022-01-30","arxiv_id":"2201.12806","repositories_listed":1,"syntology":null},{"url":"/paper/star-temporal-classification-sequence","slug":"star-temporal-classification-sequence","title":"Star Temporal Classification: Sequence Classification with Partially Labeled Data","date":"2022-01-28","arxiv_id":"2201.12208","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-phonetic-inventories-with","slug":"discovering-phonetic-inventories-with","title":"Discovering Phonetic Inventories with Crosslingual Automatic Speech Recognition","date":"2022-01-26","arxiv_id":"2201.11207","repositories_listed":1,"syntology":null},{"url":"/paper/ci-avsr-a-cantonese-audio-visual-speech","slug":"ci-avsr-a-cantonese-audio-visual-speech","title":"CI-AVSR: A Cantonese Audio-Visual Speech Dataset for In-car Command Recognition","date":"2022-01-11","arxiv_id":"2201.03804","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ci-avsr-a-cantonese-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2201.03804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.03804"}},"official":{"repos":["hltchkust/ci-avsr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"f8fceb49344359333e7a02682ab8066eb4dd3628b7678d14079ce2db0490eb93","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}