{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/7","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":65,"rows_per_page":100,"rows":[601,700],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/6","next":"/task/speech-recognition/papers/8","papers":[{"url":"/paper/diarist-streaming-speech-translation-with","slug":"diarist-streaming-speech-translation-with","title":"DiariST: Streaming Speech Translation with Speaker Diarization","date":"2023-09-14","arxiv_id":"2309.08007","repositories_listed":1,"syntology":null},{"url":"/paper/funcodec-a-fundamental-reproducible-and","slug":"funcodec-a-fundamental-reproducible-and","title":"FunCodec: A Fundamental, Reproducible and Integrable Open-source Toolkit for Neural Speech Codec","date":"2023-09-14","arxiv_id":"2309.07405","repositories_listed":1,"syntology":null},{"url":"/paper/towards-universal-speech-discrete-tokens-a","slug":"towards-universal-speech-discrete-tokens-a","title":"Towards Universal Speech Discrete Tokens: A Case Study for ASR and TTS","date":"2023-09-14","arxiv_id":"2309.07377","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-asr-for-resource-constrained-robots","slug":"hybrid-asr-for-resource-constrained-robots","title":"Hybrid ASR for Resource-Constrained Robots: HMM - Deep Learning Fusion","date":"2023-09-11","arxiv_id":"2309.07164","repositories_listed":1,"syntology":null},{"url":"/paper/active-learning-for-classifying-2d-grid-based","slug":"active-learning-for-classifying-2d-grid-based","title":"Active Learning for Classifying 2D Grid-Based Level Completability","date":"2023-09-08","arxiv_id":"2309.04367","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/active-learning-for-classifying-2d-grid-based#ran","syntology_url":"https://syntology.ai/paper/2309.04367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04367"}},"official":{"repos":["mahsabazzaz/level-completabilty-x-active-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-speech-recognition-and-disfluency-1","slug":"end-to-end-speech-recognition-and-disfluency-1","title":"End-to-End Speech Recognition and Disfluency Removal with Acoustic Language Model Pretraining","date":"2023-09-08","arxiv_id":"2309.04516","repositories_listed":1,"syntology":null},{"url":"/paper/perceptual-and-task-oriented-assessment-of-a","slug":"perceptual-and-task-oriented-assessment-of-a","title":"Perceptual and Task-Oriented Assessment of a Semantic Metric for ASR Evaluation","date":"2023-09-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/blsp-bootstrapping-language-speech-pre-1","slug":"blsp-bootstrapping-language-speech-pre-1","title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","date":"2023-09-02","arxiv_id":"2309.00916","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/blsp-bootstrapping-language-speech-pre-1#ran","syntology_url":"https://syntology.ai/paper/2309.00916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00916"}},"official":{"repos":["cwang621/blsp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-wikimedia-a-77-language-multilingual","slug":"speech-wikimedia-a-77-language-multilingual","title":"Speech Wikimedia: A 77 Language Multilingual Speech Dataset","date":"2023-08-30","arxiv_id":"2308.15710","repositories_listed":1,"syntology":null},{"url":"/paper/effect-of-attention-and-self-supervised","slug":"effect-of-attention-and-self-supervised","title":"Effect of Attention and Self-Supervised Speech Embeddings on Non-Semantic Speech Tasks","date":"2023-08-28","arxiv_id":"2308.14359","repositories_listed":1,"syntology":null},{"url":"/paper/a-small-and-fast-bert-for-chinese-medical","slug":"a-small-and-fast-bert-for-chinese-medical","title":"A Small and Fast BERT for Chinese Medical Punctuation Restoration","date":"2023-08-24","arxiv_id":"2308.12568","repositories_listed":1,"syntology":null},{"url":"/paper/an-effective-transformer-based-contextual","slug":"an-effective-transformer-based-contextual","title":"An Effective Transformer-based Contextual Model and Temporal Gate Pooling for Speaker Identification","date":"2023-08-22","arxiv_id":"2308.11241","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-risk-transducer-transducer-with","slug":"bayes-risk-transducer-transducer-with","title":"Bayes Risk Transducer: Transducer with Controllable Alignment Prediction","date":"2023-08-19","arxiv_id":"2308.10107","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-open-vocabulary-keyword-search-1","slug":"end-to-end-open-vocabulary-keyword-search-1","title":"End-to-End Open Vocabulary Keyword Search With Multilingual Neural Representations","date":"2023-08-15","arxiv_id":"2308.08027","repositories_listed":1,"syntology":null},{"url":"/paper/improving-audio-visual-speech-recognition-by","slug":"improving-audio-visual-speech-recognition-by","title":"Improving Audio-Visual Speech Recognition by Lip-Subword Correlation Based Visual Pre-training and Cross-Modal Fusion Encoder","date":"2023-08-14","arxiv_id":"2308.08488","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-emotion-recognition-with-speech","slug":"integrating-emotion-recognition-with-speech","title":"Integrating Emotion Recognition with Speech Recognition and Speaker Diarisation for Conversations","date":"2023-08-14","arxiv_id":"2308.07145","repositories_listed":1,"syntology":null},{"url":"/paper/omnidatacomposer-a-unified-data-structure-for","slug":"omnidatacomposer-a-unified-data-structure-for","title":"OmniDataComposer: A Unified Data Structure for Multimodal Data Fusion and Infinite Data Generation","date":"2023-08-08","arxiv_id":"2308.04126","repositories_listed":1,"syntology":null},{"url":"/paper/on-monotonic-aggregation-for-open-domain-qa","slug":"on-monotonic-aggregation-for-open-domain-qa","title":"On Monotonic Aggregation for Open-domain QA","date":"2023-08-08","arxiv_id":"2308.04176","repositories_listed":1,"syntology":null},{"url":"/paper/mispronunciation-detection-using-self","slug":"mispronunciation-detection-using-self","title":"Mispronunciation detection using self-supervised speech representations","date":"2023-07-30","arxiv_id":"2307.16324","repositories_listed":1,"syntology":null},{"url":"/paper/iroyinspeech-a-multi-purpose-yoruba-speech","slug":"iroyinspeech-a-multi-purpose-yoruba-speech","title":"ÌròyìnSpeech: A multi-purpose Yorùbá Speech Corpus","date":"2023-07-29","arxiv_id":"2307.16071","repositories_listed":1,"syntology":null},{"url":"/paper/turning-whisper-into-real-time-transcription","slug":"turning-whisper-into-real-time-transcription","title":"Turning Whisper into Real-Time Transcription System","date":"2023-07-27","arxiv_id":"2307.14743","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/turning-whisper-into-real-time-transcription#ran","syntology_url":"https://syntology.ai/paper/2307.14743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14743"}},"official":{"repos":["ufal/whisper_streaming"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-model-for-every-user-and-budget-label-free","slug":"a-model-for-every-user-and-budget-label-free","title":"A Model for Every User and Budget: Label-Free and Personalized Mixed-Precision Quantization","date":"2023-07-24","arxiv_id":"2307.12659","repositories_listed":1,"syntology":null},{"url":"/paper/adaptation-of-whisper-models-to-child-speech","slug":"adaptation-of-whisper-models-to-child-speech","title":"Adaptation of Whisper models to child speech recognition","date":"2023-07-24","arxiv_id":"2307.13008","repositories_listed":1,"syntology":null},{"url":"/paper/code-switched-urdu-asr-for-noisy-telephonic","slug":"code-switched-urdu-asr-for-noisy-telephonic","title":"Code-Switched Urdu ASR for Noisy Telephonic Environment using Data Centric Approach with Hybrid HMM and CNN-TDNN","date":"2023-07-24","arxiv_id":"2307.12759","repositories_listed":1,"syntology":null},{"url":"/paper/a-change-of-heart-improving-speech-emotion","slug":"a-change-of-heart-improving-speech-emotion","title":"A Change of Heart: Improving Speech Emotion Recognition through Speech-to-Text Modality Conversion","date":"2023-07-21","arxiv_id":"2307.11584","repositories_listed":1,"syntology":null},{"url":"/paper/topic-identification-for-spontaneous-speech","slug":"topic-identification-for-spontaneous-speech","title":"Topic Identification For Spontaneous Speech: Enriching Audio Features With Embedded Linguistic Information","date":"2023-07-21","arxiv_id":"2307.11450","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-dive-into-the-disparity-of-word-error","slug":"a-deep-dive-into-the-disparity-of-word-error","title":"A Deep Dive into the Disparity of Word Error Rates Across Thousands of NPTEL MOOC Videos","date":"2023-07-20","arxiv_id":"2307.10587","repositories_listed":1,"syntology":null},{"url":"/paper/oxfordvgg-submission-to-the-ego4d-av","slug":"oxfordvgg-submission-to-the-ego4d-av","title":"OxfordVGG Submission to the EGO4D AV Transcription Challenge","date":"2023-07-18","arxiv_id":"2307.09006","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-domain-sensitive-speech-recognition","slug":"zero-shot-domain-sensitive-speech-recognition","title":"Zero-shot Domain-sensitive Speech Recognition with Prompt-conditioning Fine-tuning","date":"2023-07-18","arxiv_id":"2307.10274","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-large-language-model-with-speech-for","slug":"adapting-large-language-model-with-speech-for","title":"Adapting Large Language Model with Speech for Fully Formatted End-to-End Speech Recognition","date":"2023-07-17","arxiv_id":"2307.08234","repositories_listed":1,"syntology":null},{"url":"/paper/towards-stealthy-backdoor-attacks-against","slug":"towards-stealthy-backdoor-attacks-against","title":"Towards Stealthy Backdoor Attacks against Speech Recognition via Elements of Sound","date":"2023-07-17","arxiv_id":"2307.08208","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-stealthy-backdoor-attacks-against#ran","syntology_url":"https://syntology.ai/paper/2307.08208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08208"}},"official":{"repos":["hanbocai/badspeech_soe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sumformer-a-linear-complexity-alternative-to","slug":"sumformer-a-linear-complexity-alternative-to","title":"SummaryMixing: A Linear-Complexity Alternative to Self-Attention for Speech Recognition and Understanding","date":"2023-07-12","arxiv_id":"2307.07421","repositories_listed":1,"syntology":null},{"url":"/paper/writer-adaptation-for-offline-text","slug":"writer-adaptation-for-offline-text","title":"Writer adaptation for offline text recognition: An exploration of neural network-based methods","date":"2023-07-11","arxiv_id":"2307.15071","repositories_listed":1,"syntology":null},{"url":"/paper/gammatonegram-representation-for-end-to-end","slug":"gammatonegram-representation-for-end-to-end","title":"Gammatonegram Representation for End-to-End Dysarthric Speech Processing Tasks: Speech Recognition, Speaker Identification, and Intelligibility Assessment","date":"2023-07-06","arxiv_id":"2307.03296","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-spoken-named-entity-recognition-a","slug":"exploring-spoken-named-entity-recognition-a","title":"Leveraging Cross-Lingual Transfer Learning in Spoken Named Entity Recognition Systems","date":"2023-07-03","arxiv_id":"2307.01310","repositories_listed":1,"syntology":null},{"url":"/paper/using-joint-training-speaker-encoder-with","slug":"using-joint-training-speaker-encoder-with","title":"Using joint training speaker encoder with consistency loss to achieve cross-lingual voice conversion and expressive voice conversion","date":"2023-07-01","arxiv_id":"2307.00393","repositories_listed":1,"syntology":null},{"url":"/paper/learning-delays-in-spiking-neural-networks","slug":"learning-delays-in-spiking-neural-networks","title":"Learning Delays in Spiking Neural Networks using Dilated Convolutions with Learnable Spacings","date":"2023-06-30","arxiv_id":"2306.17670","repositories_listed":1,"syntology":null},{"url":"/paper/lyricwhiz-robust-multilingual-zero-shot","slug":"lyricwhiz-robust-multilingual-zero-shot","title":"LyricWhiz: Robust Multilingual Zero-shot Lyrics Transcription by Whispering to ChatGPT","date":"2023-06-29","arxiv_id":"2306.17103","repositories_listed":1,"syntology":null},{"url":"/paper/long-term-conversation-analysis-exploring","slug":"long-term-conversation-analysis-exploring","title":"Long-term Conversation Analysis: Exploring Utility and Privacy","date":"2023-06-28","arxiv_id":"2306.16071","repositories_listed":1,"syntology":null},{"url":"/paper/a-reference-less-quality-metric-for-automatic","slug":"a-reference-less-quality-metric-for-automatic","title":"A Reference-less Quality Metric for Automatic Speech Recognition via Contrastive-Learning of a Multi-Language Model with Self-Supervision","date":"2023-06-21","arxiv_id":"2306.13114","repositories_listed":1,"syntology":null},{"url":"/paper/norefer-a-referenceless-quality-metric-for","slug":"norefer-a-referenceless-quality-metric-for","title":"NoRefER: a Referenceless Quality Metric for Automatic Speech Recognition via Semi-Supervised Language Model Fine-Tuning with Contrastive Learning","date":"2023-06-21","arxiv_id":"2306.12577","repositories_listed":1,"syntology":null},{"url":"/paper/hk-legicost-leveraging-non-verbatim","slug":"hk-legicost-leveraging-non-verbatim","title":"HK-LegiCoST: Leveraging Non-Verbatim Transcripts for Speech Translation","date":"2023-06-20","arxiv_id":"2306.11252","repositories_listed":1,"syntology":null},{"url":"/paper/rehearsal-free-online-continual-learning-for","slug":"rehearsal-free-online-continual-learning-for","title":"Rehearsal-Free Online Continual Learning for Automatic Speech Recognition","date":"2023-06-19","arxiv_id":"2306.10860","repositories_listed":1,"syntology":null},{"url":"/paper/duta-vc-a-duration-aware-typical-to-atypical","slug":"duta-vc-a-duration-aware-typical-to-atypical","title":"DuTa-VC: A Duration-aware Typical-to-atypical Voice Conversion Approach with Diffusion Probabilistic Model","date":"2023-06-18","arxiv_id":"2306.10588","repositories_listed":1,"syntology":null},{"url":"/paper/hearing-lips-in-noise-universal-viseme","slug":"hearing-lips-in-noise-universal-viseme","title":"Hearing Lips in Noise: Universal Viseme-Phoneme Mapping and Transfer for Robust Audio-Visual Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10563","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hearing-lips-in-noise-universal-viseme#ran","syntology_url":"https://syntology.ai/paper/2306.10563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10563"}},"official":{"repos":["yuchen005/univpm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mir-gan-refining-frame-level-modality","slug":"mir-gan-refining-frame-level-modality","title":"MIR-GAN: Refining Frame-Level Modality-Invariant Representations with Adversarial Network for Audio-Visual Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10567","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mir-gan-refining-frame-level-modality#ran","syntology_url":"https://syntology.ai/paper/2306.10567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10567"}},"official":{"repos":["yuchen005/mir-gan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/surt-2-0-advances-in-transducer-based-multi","slug":"surt-2-0-advances-in-transducer-based-multi","title":"SURT 2.0: Advances in Transducer-based Multi-talker Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10559","repositories_listed":1,"syntology":null},{"url":"/paper/pushing-the-limits-of-unsupervised-unit","slug":"pushing-the-limits-of-unsupervised-unit","title":"Pushing the Limits of Unsupervised Unit Discovery for SSL Speech Representation","date":"2023-06-15","arxiv_id":"2306.08920","repositories_listed":1,"syntology":null},{"url":"/paper/italic-an-italian-intent-classification","slug":"italic-an-italian-intent-classification","title":"ITALIC: An Italian Intent Classification Dataset","date":"2023-06-14","arxiv_id":"2306.08502","repositories_listed":1,"syntology":null},{"url":"/paper/towards-training-bilingual-and-code-switched","slug":"towards-training-bilingual-and-code-switched","title":"Unified model for code-switching speech recognition and language identification based on a concatenated tokenizer","date":"2023-06-14","arxiv_id":"2306.08753","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-longitudinal-chest-x-rays-and","slug":"utilizing-longitudinal-chest-x-rays-and","title":"Utilizing Longitudinal Chest X-Rays and Reports to Pre-Fill Radiology Reports","date":"2023-06-14","arxiv_id":"2306.08749","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-learning-based-audio-to-lyrics","slug":"contrastive-learning-based-audio-to-lyrics","title":"Contrastive Learning-Based Audio to Lyrics Alignment for Multiple Languages","date":"2023-06-13","arxiv_id":"2306.07744","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-training-for-low-resource","slug":"adversarial-training-for-low-resource","title":"Adversarial Training For Low-Resource Disfluency Correction","date":"2023-06-10","arxiv_id":"2306.06384","repositories_listed":1,"syntology":null},{"url":"/paper/opensr-open-modality-speech-recognition-via","slug":"opensr-open-modality-speech-recognition-via","title":"OpenSR: Open-Modality Speech Recognition via Maintaining Multi-Modality Alignment","date":"2023-06-10","arxiv_id":"2306.06410","repositories_listed":1,"syntology":null},{"url":"/paper/a-theory-of-unsupervised-speech-recognition","slug":"a-theory-of-unsupervised-speech-recognition","title":"A Theory of Unsupervised Speech Recognition","date":"2023-06-09","arxiv_id":"2306.07926","repositories_listed":1,"syntology":null},{"url":"/paper/allophant-cross-lingual-phoneme-recognition","slug":"allophant-cross-lingual-phoneme-recognition","title":"Allophant: Cross-lingual Phoneme Recognition with Articulatory Attributes","date":"2023-06-07","arxiv_id":"2306.04306","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-dysarthric-speech-recognition-using","slug":"arabic-dysarthric-speech-recognition-using","title":"Arabic Dysarthric Speech Recognition Using Adversarial and Signal-Based Augmentation","date":"2023-06-07","arxiv_id":"2306.04368","repositories_listed":1,"syntology":null},{"url":"/paper/zambezi-voice-a-multilingual-speech-corpus","slug":"zambezi-voice-a-multilingual-speech-corpus","title":"Zambezi Voice: A Multilingual Speech Corpus for Zambian Languages","date":"2023-06-07","arxiv_id":"2306.04428","repositories_listed":1,"syntology":null},{"url":"/paper/mavd-the-first-open-large-scale-mandarin","slug":"mavd-the-first-open-large-scale-mandarin","title":"MAVD: The First Open Large-Scale Mandarin Audio-Visual Dataset with Depth Information","date":"2023-06-04","arxiv_id":"2306.02263","repositories_listed":1,"syntology":null},{"url":"/paper/spellmapper-a-non-autoregressive-neural","slug":"spellmapper-a-non-autoregressive-neural","title":"SpellMapper: A non-autoregressive neural spellchecker for ASR customization with candidate retrieval based on n-gram mappings","date":"2023-06-04","arxiv_id":"2306.02317","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-pretrained-asr-models-to-low","slug":"adapting-pretrained-asr-models-to-low","title":"Advancing African-Accented Speech Recognition: Epistemic Uncertainty-Driven Data Selection for Generalizable ASR Models","date":"2023-06-03","arxiv_id":"2306.02105","repositories_listed":1,"syntology":null},{"url":"/paper/sgem-test-time-adaptation-for-automatic","slug":"sgem-test-time-adaptation-for-automatic","title":"SGEM: Test-Time Adaptation for Automatic Speech Recognition via Sequential-Level Generalized Entropy Minimization","date":"2023-06-03","arxiv_id":"2306.01981","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sgem-test-time-adaptation-for-automatic#ran","syntology_url":"https://syntology.ai/paper/2306.01981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01981"}},"official":{"repos":["drumpt/sgem"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-contextual-biasing-remain-effective-with","slug":"can-contextual-biasing-remain-effective-with","title":"Can Contextual Biasing Remain Effective with Whisper and GPT-2?","date":"2023-06-02","arxiv_id":"2306.01942","repositories_listed":1,"syntology":null},{"url":"/paper/distilxlsr-a-light-weight-cross-lingual","slug":"distilxlsr-a-light-weight-cross-lingual","title":"DistilXLSR: A Light Weight Cross-Lingual Speech Representation Model","date":"2023-06-02","arxiv_id":"2306.01303","repositories_listed":1,"syntology":null},{"url":"/paper/explainability-of-speech-recognition","slug":"explainability-of-speech-recognition","title":"Explainability of Speech Recognition Transformers via Gradient-based Attention Visualization","date":"2023-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improved-deepfake-detection-using-whisper","slug":"improved-deepfake-detection-using-whisper","title":"Improved DeepFake Detection Using Whisper Features","date":"2023-06-02","arxiv_id":"2306.01428","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improved-deepfake-detection-using-whisper#ran","syntology_url":"https://syntology.ai/paper/2306.01428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01428"}},"official":{"repos":["piotrkawa/deepfake-whisper-features"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slothspeech-denial-of-service-attack-against","slug":"slothspeech-denial-of-service-attack-against","title":"SlothSpeech: Denial-of-service Attack Against Speech Recognition Models","date":"2023-06-01","arxiv_id":"2306.00794","repositories_listed":1,"syntology":null},{"url":"/paper/perception-and-semantic-aware-regularization-1","slug":"perception-and-semantic-aware-regularization-1","title":"Perception and Semantic Aware Regularization for Sequential Confidence Calibration","date":"2023-05-31","arxiv_id":"2305.19498","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/perception-and-semantic-aware-regularization-1#ran","syntology_url":"https://syntology.ai/paper/2305.19498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19498"}},"official":{"repos":["husterpzh/pssr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-neural-networks-for-contextual-asr-with","slug":"graph-neural-networks-for-contextual-asr-with","title":"Graph Neural Networks for Contextual ASR with the Tree-Constrained Pointer Generator","date":"2023-05-30","arxiv_id":"2305.18824","repositories_listed":1,"syntology":null},{"url":"/paper/commonaccent-exploring-large-acoustic","slug":"commonaccent-exploring-large-acoustic","title":"CommonAccent: Exploring Large Acoustic Pretrained Models for Accent Classification Based on Common Voice","date":"2023-05-29","arxiv_id":"2305.18283","repositories_listed":1,"syntology":null},{"url":"/paper/hyperconformer-multi-head-hypermixer-for","slug":"hyperconformer-multi-head-hypermixer-for","title":"HyperConformer: Multi-head HyperMixer for Efficient Speech Recognition","date":"2023-05-29","arxiv_id":"2305.18281","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-granularity-gap-for-acoustic","slug":"bridging-the-granularity-gap-for-acoustic","title":"Bridging the Granularity Gap for Acoustic Modeling","date":"2023-05-27","arxiv_id":"2305.17356","repositories_listed":1,"syntology":null},{"url":"/paper/big-c-a-multimodal-multi-purpose-dataset-for","slug":"big-c-a-multimodal-multi-purpose-dataset-for","title":"BIG-C: a Multimodal Multi-Purpose Dataset for Bemba","date":"2023-05-26","arxiv_id":"2305.17202","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-characteristics-of-the-output","slug":"leveraging-characteristics-of-the-output","title":"DistriBlock: Identifying adversarial audio samples by leveraging characteristics of the output distribution","date":"2023-05-26","arxiv_id":"2305.17000","repositories_listed":1,"syntology":null},{"url":"/paper/copyne-better-contextual-asr-by-copying-named","slug":"copyne-better-contextual-asr-by-copying-named","title":"CopyNE: Better Contextual ASR by Copying Named Entities","date":"2023-05-22","arxiv_id":"2305.12839","repositories_listed":1,"syntology":null},{"url":"/paper/bat-boundary-aware-transducer-for-memory","slug":"bat-boundary-aware-transducer-for-memory","title":"BAT: Boundary aware transducer for memory-efficient and low-latency ASR","date":"2023-05-19","arxiv_id":"2305.11571","repositories_listed":1,"syntology":null},{"url":"/paper/blank-regularized-ctc-for-frame-skipping-in","slug":"blank-regularized-ctc-for-frame-skipping-in","title":"Blank-regularized CTC for Frame Skipping in Neural Transducer","date":"2023-05-19","arxiv_id":"2305.11558","repositories_listed":1,"syntology":null},{"url":"/paper/funasr-a-fundamental-end-to-end-speech","slug":"funasr-a-fundamental-end-to-end-speech","title":"FunASR: A Fundamental End-to-End Speech Recognition Toolkit","date":"2023-05-18","arxiv_id":"2305.11013","repositories_listed":1,"syntology":null},{"url":"/paper/making-more-of-little-data-improving-low","slug":"making-more-of-little-data-improving-low","title":"Making More of Little Data: Improving Low-Resource Automatic Speech Recognition Using Data Augmentation","date":"2023-05-18","arxiv_id":"2305.10951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-more-of-little-data-improving-low#ran","syntology_url":"https://syntology.ai/paper/2305.10951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10951"}},"official":{"repos":["bartelds/asr-augmentation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-the-hidden-talent-of-web-scale","slug":"prompting-the-hidden-talent-of-web-scale","title":"Prompting the Hidden Talent of Web-Scale Speech Models for Zero-Shot Task Generalization","date":"2023-05-18","arxiv_id":"2305.11095","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-fine-tuning-for-improved","slug":"self-supervised-fine-tuning-for-improved","title":"Self-supervised Fine-tuning for Improved Content Representations by Speaker-invariant Clustering","date":"2023-05-18","arxiv_id":"2305.11072","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-fine-tuning-for-improved#ran","syntology_url":"https://syntology.ai/paper/2305.11072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11072"}},"official":{"repos":["vectominist/spin"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speechgpt-empowering-large-language-models","slug":"speechgpt-empowering-large-language-models","title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities","date":"2023-05-18","arxiv_id":"2305.11000","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speechgpt-empowering-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2305.11000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11000"}},"official":{"repos":["0nutation/speechgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-global-interaction-and-local","slug":"cross-modal-global-interaction-and-local","title":"Cross-Modal Global Interaction and Local Alignment for Audio-Visual Speech Recognition","date":"2023-05-16","arxiv_id":"2305.09212","repositories_listed":1,"syntology":null},{"url":"/paper/back-translation-for-speech-to-text","slug":"back-translation-for-speech-to-text","title":"Back Translation for Speech-to-text Translation Without Transcripts","date":"2023-05-15","arxiv_id":"2305.08709","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-variants-of-wav2vec-2-0-on","slug":"evaluating-variants-of-wav2vec-2-0-on","title":"Evaluating Variants of wav2vec 2.0 on Affective Vocal Burst Tasks","date":"2023-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-domain-adaptation-for-self","slug":"towards-better-domain-adaptation-for-self","title":"Towards Better Domain Adaptation for Self-supervised Models: A Case Study of Child ASR","date":"2023-04-28","arxiv_id":"2305.00115","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-deep-learning-models-for-raspberry","slug":"optimizing-deep-learning-models-for-raspberry","title":"Optimizing Deep Learning Models For Raspberry Pi","date":"2023-04-25","arxiv_id":"2304.13039","repositories_listed":1,"syntology":null},{"url":"/paper/olisia-a-cascade-system-for-spoken-dialogue","slug":"olisia-a-cascade-system-for-spoken-dialogue","title":"OLISIA: a Cascade System for Spoken Dialogue State Tracking","date":"2023-04-20","arxiv_id":"2304.11073","repositories_listed":1,"syntology":null},{"url":"/paper/cb-conformer-contextual-biasing-conformer-for","slug":"cb-conformer-contextual-biasing-conformer-for","title":"CB-Conformer: Contextual biasing Conformer for biased word recognition","date":"2023-04-19","arxiv_id":"2304.09607","repositories_listed":1,"syntology":null},{"url":"/paper/political-corpus-creation-through-automatic","slug":"political-corpus-creation-through-automatic","title":"Political corpus creation through automatic speech recognition on EU debates","date":"2023-04-17","arxiv_id":"2304.08137","repositories_listed":1,"syntology":null},{"url":"/paper/acoustic-absement-in-detail-quantifying","slug":"acoustic-absement-in-detail-quantifying","title":"Acoustic absement in detail: Quantifying acoustic differences across time-series representations of speech data","date":"2023-04-12","arxiv_id":"2304.06183","repositories_listed":1,"syntology":null},{"url":"/paper/certifiable-black-box-attack-ensuring","slug":"certifiable-black-box-attack-ensuring","title":"Certifiable Black-Box Attacks with Randomized Adversarial Examples: Breaking Defenses with Provable Confidence","date":"2023-04-10","arxiv_id":"2304.04343","repositories_listed":1,"syntology":null},{"url":"/paper/cico-domain-aware-sign-language-retrieval-via","slug":"cico-domain-aware-sign-language-retrieval-via","title":"CiCo: Domain-Aware Sign Language Retrieval via Cross-Lingual Contrastive Learning","date":"2023-03-22","arxiv_id":"2303.12793","repositories_listed":1,"syntology":null},{"url":"/paper/cascading-and-direct-approaches-to","slug":"cascading-and-direct-approaches-to","title":"Cascading and Direct Approaches to Unsupervised Constituency Parsing on Spoken Sentences","date":"2023-03-15","arxiv_id":"2303.08809","repositories_listed":1,"syntology":null},{"url":"/paper/hybridformer-improving-squeezeformer-with","slug":"hybridformer-improving-squeezeformer-with","title":"HYBRIDFORMER: improving SqueezeFormer with hybrid attention and NSR mechanism","date":"2023-03-15","arxiv_id":"2303.08636","repositories_listed":1,"syntology":null},{"url":"/paper/watch-or-listen-robust-audio-visual-speech","slug":"watch-or-listen-robust-audio-visual-speech","title":"Watch or Listen: Robust Audio-Visual Speech Recognition with Visual Corruption Modeling and Reliability Scoring","date":"2023-03-15","arxiv_id":"2303.08536","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/watch-or-listen-robust-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2303.08536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08536"}},"official":{"repos":["ms-dot-k/AVSR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/i3d-transformer-architectures-with-input","slug":"i3d-transformer-architectures-with-input","title":"I3D: Transformer architectures with input-dependent dynamic depth for speech recognition","date":"2023-03-14","arxiv_id":"2303.07624","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-strategies-for-faster-inference","slug":"fine-tuning-strategies-for-faster-inference","title":"Fine-tuning Strategies for Faster Inference using Speech Self-Supervised Models: A Comparative Study","date":"2023-03-12","arxiv_id":"2303.06740","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-transformer-training-by","slug":"stabilizing-transformer-training-by","title":"Stabilizing Transformer Training by Preventing Attention Entropy Collapse","date":"2023-03-11","arxiv_id":"2303.06296","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformer-training-by#ran","syntology_url":"https://syntology.ai/paper/2303.06296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06296"}},"official":{"repos":["apple/ml-sigma-reparam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transcription-free-filler-word-detection-with","slug":"transcription-free-filler-word-detection-with","title":"Transcription free filler word detection with Neural semi-CRFs","date":"2023-03-11","arxiv_id":"2303.06475","repositories_listed":1,"syntology":null}],"record_sha256":"57e1fc5fe723c1997ceb11b8b4d917121dfe2e2307397fb176db113193a2706f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}