{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/3","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":31,"rows_per_page":100,"rows":[201,300],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition/papers/2","next":"/task/automatic-speech-recognition/papers/4","papers":[{"url":"/paper/multilingual-speech-models-for-automatic","slug":"multilingual-speech-models-for-automatic","title":"Twists, Humps, and Pebbles: Multilingual Speech Recognition Models Exhibit Gender Performance Gaps","date":"2024-02-28","arxiv_id":"2402.17954","repositories_listed":1,"syntology":null},{"url":"/paper/owsm-ctc-an-open-encoder-only-speech","slug":"owsm-ctc-an-open-encoder-only-speech","title":"OWSM-CTC: An Open Encoder-Only Speech Foundation Model for Speech Recognition, Translation, and Language Identification","date":"2024-02-20","arxiv_id":"2402.12654","repositories_listed":1,"syntology":null},{"url":"/paper/air-bench-benchmarking-large-audio-language","slug":"air-bench-benchmarking-large-audio-language","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","date":"2024-02-12","arxiv_id":"2402.07729","repositories_listed":1,"syntology":null},{"url":"/paper/it-s-never-too-late-fusing-acoustic","slug":"it-s-never-too-late-fusing-acoustic","title":"It's Never Too Late: Fusing Acoustic Information into Large Language Models for Automatic Speech Recognition","date":"2024-02-08","arxiv_id":"2402.05457","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/it-s-never-too-late-fusing-acoustic#ran","syntology_url":"https://syntology.ai/paper/2402.05457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05457"}},"official":null}},{"url":"/paper/unified-speech-text-pretraining-for-spoken","slug":"unified-speech-text-pretraining-for-spoken","title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","date":"2024-02-08","arxiv_id":"2402.05706","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":5,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unified-speech-text-pretraining-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2402.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05706"}},"official":{"repos":["naver-ai/usdm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reborn-reinforcement-learned-boundary","slug":"reborn-reinforcement-learned-boundary","title":"REBORN: Reinforcement-Learned Boundary Segmentation with Iterative Training for Unsupervised ASR","date":"2024-02-06","arxiv_id":"2402.03988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/reborn-reinforcement-learned-boundary#ran","syntology_url":"https://syntology.ai/paper/2402.03988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03988"}},"official":{"repos":["andybi7676/reborn-uasr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-sequence-transduction-through","slug":"streaming-sequence-transduction-through","title":"Streaming Sequence Transduction through Dynamic Compression","date":"2024-02-02","arxiv_id":"2402.01172","repositories_listed":1,"syntology":null},{"url":"/paper/word-level-asr-quality-estimation-for","slug":"word-level-asr-quality-estimation-for","title":"Word-Level ASR Quality Estimation for Efficient Corpus Sampling and Post-Editing through Analyzing Attentions of a Reference-Free Metric","date":"2024-01-20","arxiv_id":"2401.11268","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-efficient-learners","slug":"large-language-models-are-efficient-learners","title":"Large Language Models are Efficient Learners of Noise-Robust Speech Recognition","date":"2024-01-19","arxiv_id":"2401.10446","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-efficient-learners#ran","syntology_url":"https://syntology.ai/paper/2401.10446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10446"}},"official":{"repos":["yuchen005/robustger"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cascaded-cross-modal-transformer-for-audio","slug":"cascaded-cross-modal-transformer-for-audio","title":"Cascaded Cross-Modal Transformer for Audio-Textual Classification","date":"2024-01-15","arxiv_id":"2401.07575","repositories_listed":1,"syntology":null},{"url":"/paper/multichannel-av-wav2vec2-a-framework-for","slug":"multichannel-av-wav2vec2-a-framework-for","title":"Multichannel AV-wav2vec2: A Framework for Learning Multichannel Multi-Modal Speech Representation","date":"2024-01-07","arxiv_id":"2401.03468","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":2,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multichannel-av-wav2vec2-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2401.03468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03468"}},"official":{"repos":["zqs01/multi-channel-wav2vec2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/teles-temporal-lexeme-similarity-score-to","slug":"teles-temporal-lexeme-similarity-score-to","title":"TeLeS: Temporal Lexeme Similarity Score to Estimate Confidence in End-to-End ASR","date":"2024-01-06","arxiv_id":"2401.03251","repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-dialogue-as-a-catalyst-for-self","slug":"task-oriented-dialogue-as-a-catalyst-for-self","title":"Task Oriented Dialogue as a Catalyst for Self-Supervised Automatic Speech Recognition","date":"2024-01-04","arxiv_id":"2401.02417","repositories_listed":1,"syntology":null},{"url":"/paper/stable-distillation-regularizing-continued","slug":"stable-distillation-regularizing-continued","title":"Stable Distillation: Regularizing Continued Pre-training for Low-Resource Automatic Speech Recognition","date":"2023-12-20","arxiv_id":"2312.12783","repositories_listed":1,"syntology":null},{"url":"/paper/seq2seq-for-automatic-paraphasia-detection-in","slug":"seq2seq-for-automatic-paraphasia-detection-in","title":"Seq2seq for Automatic Paraphasia Detection in Aphasic Speech","date":"2023-12-16","arxiv_id":"2312.10518","repositories_listed":1,"syntology":null},{"url":"/paper/extending-whisper-with-prompt-tuning-to","slug":"extending-whisper-with-prompt-tuning-to","title":"Extending Whisper with prompt tuning to target-speaker ASR","date":"2023-12-13","arxiv_id":"2312.08079","repositories_listed":1,"syntology":null},{"url":"/paper/rose-a-recognition-oriented-speech","slug":"rose-a-recognition-oriented-speech","title":"ROSE: A Recognition-Oriented Speech Enhancement Framework in Air Traffic Control Using Multi-Objective Learning","date":"2023-12-11","arxiv_id":"2312.06118","repositories_listed":1,"syntology":null},{"url":"/paper/bigger-is-not-always-better-the-effect-of","slug":"bigger-is-not-always-better-the-effect-of","title":"Bigger is not Always Better: The Effect of Context Size on Speech Pre-Training","date":"2023-12-03","arxiv_id":"2312.01515","repositories_listed":1,"syntology":null},{"url":"/paper/d4am-a-general-denoising-framework-for","slug":"d4am-a-general-denoising-framework-for","title":"D4AM: A General Denoising Framework for Downstream Acoustic Models","date":"2023-11-28","arxiv_id":"2311.16595","repositories_listed":1,"syntology":null},{"url":"/paper/a-quantitative-approach-to-understand-self","slug":"a-quantitative-approach-to-understand-self","title":"A Quantitative Approach to Understand Self-Supervised Models as Cross-lingual Feature Extractors","date":"2023-11-27","arxiv_id":"2311.15954","repositories_listed":1,"syntology":null},{"url":"/paper/improving-whispered-speech-recognition","slug":"improving-whispered-speech-recognition","title":"Improving Whispered Speech Recognition Performance using Pseudo-whispered based Data Augmentation","date":"2023-11-09","arxiv_id":"2311.05179","repositories_listed":1,"syntology":null},{"url":"/paper/improved-child-text-to-speech-synthesis","slug":"improved-child-text-to-speech-synthesis","title":"Improved Child Text-to-Speech Synthesis through Fastpitch-based Transfer Learning","date":"2023-11-07","arxiv_id":"2311.04313","repositories_listed":1,"syntology":null},{"url":"/paper/pseudo-labeling-for-domain-agnostic-bangla","slug":"pseudo-labeling-for-domain-agnostic-bangla","title":"Pseudo-Labeling for Domain-Agnostic Bangla Automatic Speech Recognition","date":"2023-11-06","arxiv_id":"2311.03196","repositories_listed":1,"syntology":null},{"url":"/paper/distilwhisper-efficient-distillation-of-multi","slug":"distilwhisper-efficient-distillation-of-multi","title":"Multilingual DistilWhisper: Efficient Distillation of Multi-task Speech Models via Language-Specific Experts","date":"2023-11-02","arxiv_id":"2311.01070","repositories_listed":1,"syntology":null},{"url":"/paper/whisper-mce-whisper-model-finetuned-for","slug":"whisper-mce-whisper-model-finetuned-for","title":"Developing a Multilingual Dataset and Evaluation Metrics for Code-Switching: A Focus on Hong Kong's Polylingual Dynamics","date":"2023-10-27","arxiv_id":"2310.17953","repositories_listed":1,"syntology":null},{"url":"/paper/artst-arabic-text-and-speech-transformer","slug":"artst-arabic-text-and-speech-transformer","title":"ArTST: Arabic Text and Speech Transformer","date":"2023-10-25","arxiv_id":"2310.16621","repositories_listed":1,"syntology":null},{"url":"/paper/cl-masr-a-continual-learning-benchmark-for","slug":"cl-masr-a-continual-learning-benchmark-for","title":"CL-MASR: A Continual Learning Benchmark for Multilingual ASR","date":"2023-10-25","arxiv_id":"2310.16931","repositories_listed":1,"syntology":null},{"url":"/paper/disco-a-large-scale-human-annotated-corpus","slug":"disco-a-large-scale-human-annotated-corpus","title":"DISCO: A Large Scale Human Annotated Corpus for Disfluency Correction in Indo-European Languages","date":"2023-10-25","arxiv_id":"2310.16749","repositories_listed":1,"syntology":null},{"url":"/paper/accented-speech-recognition-with-accent","slug":"accented-speech-recognition-with-accent","title":"Accented Speech Recognition With Accent-specific Codebooks","date":"2023-10-24","arxiv_id":"2310.15970","repositories_listed":1,"syntology":null},{"url":"/paper/zipformer-a-faster-and-better-encoder-for","slug":"zipformer-a-faster-and-better-encoder-for","title":"Zipformer: A faster and better encoder for automatic speech recognition","date":"2023-10-17","arxiv_id":"2310.11230","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-test-time-adaptation-for-acoustic","slug":"advancing-test-time-adaptation-for-acoustic","title":"Advancing Test-Time Adaptation in Wild Acoustic Test Settings","date":"2023-10-14","arxiv_id":"2310.09505","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-test-time-adaptation-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2310.09505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09505"}},"official":{"repos":["Waffle-Liu/CEA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/salm-speech-augmented-language-model-with-in","slug":"salm-speech-augmented-language-model-with-in","title":"SALM: Speech-augmented Language Model with In-context Learning for Speech Recognition and Translation","date":"2023-10-13","arxiv_id":"2310.09424","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-the-adapters-for-code-switching-in","slug":"adapting-the-adapters-for-code-switching-in","title":"Adapting the adapters for code-switching in multilingual ASR","date":"2023-10-11","arxiv_id":"2310.07423","repositories_listed":1,"syntology":null},{"url":"/paper/no-pitch-left-behind-addressing-gender","slug":"no-pitch-left-behind-addressing-gender","title":"No Pitch Left Behind: Addressing Gender Unbalance in Automatic Speech Recognition through Pitch Manipulation","date":"2023-10-10","arxiv_id":"2310.06590","repositories_listed":1,"syntology":null},{"url":"/paper/whispering-llama-a-cross-modal-generative","slug":"whispering-llama-a-cross-modal-generative","title":"Whispering LLaMA: A Cross-Modal Generative Error Correction Framework for Speech Recognition","date":"2023-10-10","arxiv_id":"2310.06434","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/whispering-llama-a-cross-modal-generative#ran","syntology_url":"https://syntology.ai/paper/2310.06434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06434"}},"official":{"repos":["srijith-rkr/whispering-llama"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ed-cec-improving-rare-word-recognition-using","slug":"ed-cec-improving-rare-word-recognition-using","title":"ed-cec: improving rare word recognition using asr postprocessing based on error detection and context-aware error correction","date":"2023-10-08","arxiv_id":"2310.05129","repositories_listed":1,"syntology":null},{"url":"/paper/hyporadise-an-open-baseline-for-generative-1","slug":"hyporadise-an-open-baseline-for-generative-1","title":"HyPoradise: An Open Baseline for Generative Speech Recognition with Large Language Models","date":"2023-09-27","arxiv_id":"2309.15701","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyporadise-an-open-baseline-for-generative-1#ran","syntology_url":"https://syntology.ai/paper/2309.15701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15701"}},"official":{"repos":["hypotheses-paradise/hypo2trans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-collage-code-switched-audio-generation","slug":"speech-collage-code-switched-audio-generation","title":"Speech collage: code-switched audio generation by collaging monolingual corpora","date":"2023-09-27","arxiv_id":"2309.15674","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-flawed-data-weakly-supervised","slug":"learning-from-flawed-data-weakly-supervised","title":"Learning from Flawed Data: Weakly Supervised Automatic Speech Recognition","date":"2023-09-26","arxiv_id":"2309.15796","repositories_listed":1,"syntology":null},{"url":"/paper/segmentation-free-streaming-machine","slug":"segmentation-free-streaming-machine","title":"Segmentation-Free Streaming Machine Translation","date":"2023-09-26","arxiv_id":"2309.14823","repositories_listed":1,"syntology":null},{"url":"/paper/human-transcription-quality-improvement","slug":"human-transcription-quality-improvement","title":"Human Transcription Quality Improvement","date":"2023-09-24","arxiv_id":"2309.14372","repositories_listed":1,"syntology":null},{"url":"/paper/big-model-only-for-hard-audios-sample","slug":"big-model-only-for-hard-audios-sample","title":"Big model only for hard audios: Sample dependent Whisper model selection for efficient inferences","date":"2023-09-22","arxiv_id":"2309.12712","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-conformer-for-improved-end","slug":"memory-augmented-conformer-for-improved-end","title":"Memory-augmented conformer for improved end-to-end long-form ASR","date":"2023-09-22","arxiv_id":"2309.13029","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-the-zero-shot-power-of-instruction","slug":"harnessing-the-zero-shot-power-of-instruction","title":"Harnessing the Zero-Shot Power of Instruction-Tuned Large Language Model in End-to-End Speech Recognition","date":"2023-09-19","arxiv_id":"2309.10524","repositories_listed":1,"syntology":null},{"url":"/paper/hypr-a-comprehensive-study-for-asr-hypothesis","slug":"hypr-a-comprehensive-study-for-asr-hypothesis","title":"HypR: A comprehensive study for ASR hypothesis revising with a reference corpus","date":"2023-09-18","arxiv_id":"2309.09838","repositories_listed":1,"syntology":null},{"url":"/paper/training-dynamic-models-using-early-exits-for","slug":"training-dynamic-models-using-early-exits-for","title":"Training dynamic models using early exits for automatic speech recognition on resource-constrained devices","date":"2023-09-18","arxiv_id":"2309.09546","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-quantised-end-to-end-asr-models-via","slug":"enhancing-quantised-end-to-end-asr-models-via","title":"Enhancing Quantised End-to-End ASR Models via Personalisation","date":"2023-09-17","arxiv_id":"2309.09136","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-punctuation-restoration-for","slug":"transformer-based-punctuation-restoration-for","title":"Transformer Based Punctuation Restoration for Turkish","date":"2023-09-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-asr-for-resource-constrained-robots","slug":"hybrid-asr-for-resource-constrained-robots","title":"Hybrid ASR for Resource-Constrained Robots: HMM - Deep Learning Fusion","date":"2023-09-11","arxiv_id":"2309.07164","repositories_listed":1,"syntology":null},{"url":"/paper/perceptual-and-task-oriented-assessment-of-a","slug":"perceptual-and-task-oriented-assessment-of-a","title":"Perceptual and Task-Oriented Assessment of a Semantic Metric for ASR Evaluation","date":"2023-09-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-small-and-fast-bert-for-chinese-medical","slug":"a-small-and-fast-bert-for-chinese-medical","title":"A Small and Fast BERT for Chinese Medical Punctuation Restoration","date":"2023-08-24","arxiv_id":"2308.12568","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-risk-transducer-transducer-with","slug":"bayes-risk-transducer-transducer-with","title":"Bayes Risk Transducer: Transducer with Controllable Alignment Prediction","date":"2023-08-19","arxiv_id":"2308.10107","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-open-vocabulary-keyword-search-1","slug":"end-to-end-open-vocabulary-keyword-search-1","title":"End-to-End Open Vocabulary Keyword Search With Multilingual Neural Representations","date":"2023-08-15","arxiv_id":"2308.08027","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-emotion-recognition-with-speech","slug":"integrating-emotion-recognition-with-speech","title":"Integrating Emotion Recognition with Speech Recognition and Speaker Diarisation for Conversations","date":"2023-08-14","arxiv_id":"2308.07145","repositories_listed":1,"syntology":null},{"url":"/paper/omnidatacomposer-a-unified-data-structure-for","slug":"omnidatacomposer-a-unified-data-structure-for","title":"OmniDataComposer: A Unified Data Structure for Multimodal Data Fusion and Infinite Data Generation","date":"2023-08-08","arxiv_id":"2308.04126","repositories_listed":1,"syntology":null},{"url":"/paper/iroyinspeech-a-multi-purpose-yoruba-speech","slug":"iroyinspeech-a-multi-purpose-yoruba-speech","title":"ÌròyìnSpeech: A multi-purpose Yorùbá Speech Corpus","date":"2023-07-29","arxiv_id":"2307.16071","repositories_listed":1,"syntology":null},{"url":"/paper/a-model-for-every-user-and-budget-label-free","slug":"a-model-for-every-user-and-budget-label-free","title":"A Model for Every User and Budget: Label-Free and Personalized Mixed-Precision Quantization","date":"2023-07-24","arxiv_id":"2307.12659","repositories_listed":1,"syntology":null},{"url":"/paper/adaptation-of-whisper-models-to-child-speech","slug":"adaptation-of-whisper-models-to-child-speech","title":"Adaptation of Whisper models to child speech recognition","date":"2023-07-24","arxiv_id":"2307.13008","repositories_listed":1,"syntology":null},{"url":"/paper/a-change-of-heart-improving-speech-emotion","slug":"a-change-of-heart-improving-speech-emotion","title":"A Change of Heart: Improving Speech Emotion Recognition through Speech-to-Text Modality Conversion","date":"2023-07-21","arxiv_id":"2307.11584","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-dive-into-the-disparity-of-word-error","slug":"a-deep-dive-into-the-disparity-of-word-error","title":"A Deep Dive into the Disparity of Word Error Rates Across Thousands of NPTEL MOOC Videos","date":"2023-07-20","arxiv_id":"2307.10587","repositories_listed":1,"syntology":null},{"url":"/paper/a-reference-less-quality-metric-for-automatic","slug":"a-reference-less-quality-metric-for-automatic","title":"A Reference-less Quality Metric for Automatic Speech Recognition via Contrastive-Learning of a Multi-Language Model with Self-Supervision","date":"2023-06-21","arxiv_id":"2306.13114","repositories_listed":1,"syntology":null},{"url":"/paper/norefer-a-referenceless-quality-metric-for","slug":"norefer-a-referenceless-quality-metric-for","title":"NoRefER: a Referenceless Quality Metric for Automatic Speech Recognition via Semi-Supervised Language Model Fine-Tuning with Contrastive Learning","date":"2023-06-21","arxiv_id":"2306.12577","repositories_listed":1,"syntology":null},{"url":"/paper/rehearsal-free-online-continual-learning-for","slug":"rehearsal-free-online-continual-learning-for","title":"Rehearsal-Free Online Continual Learning for Automatic Speech Recognition","date":"2023-06-19","arxiv_id":"2306.10860","repositories_listed":1,"syntology":null},{"url":"/paper/towards-training-bilingual-and-code-switched","slug":"towards-training-bilingual-and-code-switched","title":"Unified model for code-switching speech recognition and language identification based on a concatenated tokenizer","date":"2023-06-14","arxiv_id":"2306.08753","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-training-for-low-resource","slug":"adversarial-training-for-low-resource","title":"Adversarial Training For Low-Resource Disfluency Correction","date":"2023-06-10","arxiv_id":"2306.06384","repositories_listed":1,"syntology":null},{"url":"/paper/a-theory-of-unsupervised-speech-recognition","slug":"a-theory-of-unsupervised-speech-recognition","title":"A Theory of Unsupervised Speech Recognition","date":"2023-06-09","arxiv_id":"2306.07926","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-dysarthric-speech-recognition-using","slug":"arabic-dysarthric-speech-recognition-using","title":"Arabic Dysarthric Speech Recognition Using Adversarial and Signal-Based Augmentation","date":"2023-06-07","arxiv_id":"2306.04368","repositories_listed":1,"syntology":null},{"url":"/paper/spellmapper-a-non-autoregressive-neural","slug":"spellmapper-a-non-autoregressive-neural","title":"SpellMapper: A non-autoregressive neural spellchecker for ASR customization with candidate retrieval based on n-gram mappings","date":"2023-06-04","arxiv_id":"2306.02317","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-pretrained-asr-models-to-low","slug":"adapting-pretrained-asr-models-to-low","title":"Advancing African-Accented Speech Recognition: Epistemic Uncertainty-Driven Data Selection for Generalizable ASR Models","date":"2023-06-03","arxiv_id":"2306.02105","repositories_listed":1,"syntology":null},{"url":"/paper/sgem-test-time-adaptation-for-automatic","slug":"sgem-test-time-adaptation-for-automatic","title":"SGEM: Test-Time Adaptation for Automatic Speech Recognition via Sequential-Level Generalized Entropy Minimization","date":"2023-06-03","arxiv_id":"2306.01981","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sgem-test-time-adaptation-for-automatic#ran","syntology_url":"https://syntology.ai/paper/2306.01981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01981"}},"official":{"repos":["drumpt/sgem"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-contextual-biasing-remain-effective-with","slug":"can-contextual-biasing-remain-effective-with","title":"Can Contextual Biasing Remain Effective with Whisper and GPT-2?","date":"2023-06-02","arxiv_id":"2306.01942","repositories_listed":1,"syntology":null},{"url":"/paper/explainability-of-speech-recognition","slug":"explainability-of-speech-recognition","title":"Explainability of Speech Recognition Transformers via Gradient-based Attention Visualization","date":"2023-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/slothspeech-denial-of-service-attack-against","slug":"slothspeech-denial-of-service-attack-against","title":"SlothSpeech: Denial-of-service Attack Against Speech Recognition Models","date":"2023-06-01","arxiv_id":"2306.00794","repositories_listed":1,"syntology":null},{"url":"/paper/graph-neural-networks-for-contextual-asr-with","slug":"graph-neural-networks-for-contextual-asr-with","title":"Graph Neural Networks for Contextual ASR with the Tree-Constrained Pointer Generator","date":"2023-05-30","arxiv_id":"2305.18824","repositories_listed":1,"syntology":null},{"url":"/paper/commonaccent-exploring-large-acoustic","slug":"commonaccent-exploring-large-acoustic","title":"CommonAccent: Exploring Large Acoustic Pretrained Models for Accent Classification Based on Common Voice","date":"2023-05-29","arxiv_id":"2305.18283","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-characteristics-of-the-output","slug":"leveraging-characteristics-of-the-output","title":"DistriBlock: Identifying adversarial audio samples by leveraging characteristics of the output distribution","date":"2023-05-26","arxiv_id":"2305.17000","repositories_listed":1,"syntology":null},{"url":"/paper/copyne-better-contextual-asr-by-copying-named","slug":"copyne-better-contextual-asr-by-copying-named","title":"CopyNE: Better Contextual ASR by Copying Named Entities","date":"2023-05-22","arxiv_id":"2305.12839","repositories_listed":1,"syntology":null},{"url":"/paper/bat-boundary-aware-transducer-for-memory","slug":"bat-boundary-aware-transducer-for-memory","title":"BAT: Boundary aware transducer for memory-efficient and low-latency ASR","date":"2023-05-19","arxiv_id":"2305.11571","repositories_listed":1,"syntology":null},{"url":"/paper/making-more-of-little-data-improving-low","slug":"making-more-of-little-data-improving-low","title":"Making More of Little Data: Improving Low-Resource Automatic Speech Recognition Using Data Augmentation","date":"2023-05-18","arxiv_id":"2305.10951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-more-of-little-data-improving-low#ran","syntology_url":"https://syntology.ai/paper/2305.10951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10951"}},"official":{"repos":["bartelds/asr-augmentation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-global-interaction-and-local","slug":"cross-modal-global-interaction-and-local","title":"Cross-Modal Global Interaction and Local Alignment for Audio-Visual Speech Recognition","date":"2023-05-16","arxiv_id":"2305.09212","repositories_listed":1,"syntology":null},{"url":"/paper/back-translation-for-speech-to-text","slug":"back-translation-for-speech-to-text","title":"Back Translation for Speech-to-text Translation Without Transcripts","date":"2023-05-15","arxiv_id":"2305.08709","repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-domain-adaptation-for-self","slug":"towards-better-domain-adaptation-for-self","title":"Towards Better Domain Adaptation for Self-supervised Models: A Case Study of Child ASR","date":"2023-04-28","arxiv_id":"2305.00115","repositories_listed":1,"syntology":null},{"url":"/paper/olisia-a-cascade-system-for-spoken-dialogue","slug":"olisia-a-cascade-system-for-spoken-dialogue","title":"OLISIA: a Cascade System for Spoken Dialogue State Tracking","date":"2023-04-20","arxiv_id":"2304.11073","repositories_listed":1,"syntology":null},{"url":"/paper/political-corpus-creation-through-automatic","slug":"political-corpus-creation-through-automatic","title":"Political corpus creation through automatic speech recognition on EU debates","date":"2023-04-17","arxiv_id":"2304.08137","repositories_listed":1,"syntology":null},{"url":"/paper/cascading-and-direct-approaches-to","slug":"cascading-and-direct-approaches-to","title":"Cascading and Direct Approaches to Unsupervised Constituency Parsing on Spoken Sentences","date":"2023-03-15","arxiv_id":"2303.08809","repositories_listed":1,"syntology":null},{"url":"/paper/hybridformer-improving-squeezeformer-with","slug":"hybridformer-improving-squeezeformer-with","title":"HYBRIDFORMER: improving SqueezeFormer with hybrid attention and NSR mechanism","date":"2023-03-15","arxiv_id":"2303.08636","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-strategies-for-faster-inference","slug":"fine-tuning-strategies-for-faster-inference","title":"Fine-tuning Strategies for Faster Inference using Speech Self-Supervised Models: A Comparative Study","date":"2023-03-12","arxiv_id":"2303.06740","repositories_listed":1,"syntology":null},{"url":"/paper/transcription-free-filler-word-detection-with","slug":"transcription-free-filler-word-detection-with","title":"Transcription free filler word detection with Neural semi-CRFs","date":"2023-03-11","arxiv_id":"2303.06475","repositories_listed":1,"syntology":null},{"url":"/paper/language-universal-adapter-learning-with","slug":"language-universal-adapter-learning-with","title":"Language-Universal Adapter Learning with Knowledge Distillation for End-to-End Multilingual Speech Recognition","date":"2023-02-28","arxiv_id":"2303.01249","repositories_listed":1,"syntology":null},{"url":"/paper/low-latency-transformers-for-speech","slug":"low-latency-transformers-for-speech","title":"A low latency attention module for streaming self-supervised speech representation learning","date":"2023-02-27","arxiv_id":"2302.13451","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-speech-recognition-for-language","slug":"multimodal-speech-recognition-for-language","title":"Multimodal Speech Recognition for Language-Guided Embodied Agents","date":"2023-02-27","arxiv_id":"2302.14030","repositories_listed":1,"syntology":null},{"url":"/paper/text-only-domain-adaptation-for-end-to-end","slug":"text-only-domain-adaptation-for-end-to-end","title":"Text-only domain adaptation for end-to-end ASR using integrated text-to-mel-spectrogram generator","date":"2023-02-27","arxiv_id":"2302.14036","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-ensemble-architecture-for","slug":"efficient-ensemble-architecture-for","title":"Efficient Ensemble for Multimodal Punctuation Restoration using Time-Delay Neural Network","date":"2023-02-26","arxiv_id":"2302.13376","repositories_listed":1,"syntology":null},{"url":"/paper/improving-massively-multilingual-asr-with","slug":"improving-massively-multilingual-asr-with","title":"Improving Massively Multilingual ASR With Auxiliary CTC Objectives","date":"2023-02-24","arxiv_id":"2302.12829","repositories_listed":1,"syntology":null},{"url":"/paper/gradient-remedy-for-multi-task-learning-in","slug":"gradient-remedy-for-multi-task-learning-in","title":"Gradient Remedy for Multi-Task Learning in End-to-End Noise-Robust Speech Recognition","date":"2023-02-22","arxiv_id":"2302.11362","repositories_listed":1,"syntology":null},{"url":"/paper/a-sidecar-separator-can-convert-a-single","slug":"a-sidecar-separator-can-convert-a-single","title":"A Sidecar Separator Can Convert a Single-Talker Speech Recognition System to a Multi-Talker One","date":"2023-02-20","arxiv_id":"2302.09908","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-score-based-speaker-adaptation-of","slug":"confidence-score-based-speaker-adaptation-of","title":"Confidence Score Based Speaker Adaptation of Conformer Speech Recognition Systems","date":"2023-02-15","arxiv_id":"2302.07521","repositories_listed":1,"syntology":null},{"url":"/paper/asdf-a-differential-testing-framework-for","slug":"asdf-a-differential-testing-framework-for","title":"ASDF: A Differential Testing Framework for Automatic Speech Recognition Systems","date":"2023-02-11","arxiv_id":"2302.05582","repositories_listed":1,"syntology":null},{"url":"/paper/complex-dynamic-neurons-improved-spiking","slug":"complex-dynamic-neurons-improved-spiking","title":"Complex Dynamic Neurons Improved Spiking Transformer Network for Efficient Automatic Speech Recognition","date":"2023-02-02","arxiv_id":"2302.01194","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/complex-dynamic-neurons-improved-spiking#ran","syntology_url":"https://syntology.ai/paper/2302.01194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01194"}},"official":{"repos":["MingLunHan/CIF-PyTorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/syllable-subword-tokens-for-open-vocabulary","slug":"syllable-subword-tokens-for-open-vocabulary","title":"Syllable Subword Tokens for Open Vocabulary Speech Recognition in Malayalam","date":"2023-01-17","arxiv_id":"2301.06736","repositories_listed":1,"syntology":null}],"record_sha256":"1f99503123325b4d01501889354e8b0fbb469f609c86e1d45548eea5da977f54","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}