{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition/papers/2","list_of":"/task/automatic-speech-recognition","task":"Automatic Speech Recognition (ASR)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":31,"rows_per_page":100,"rows":[101,200],"of":3012,"counts":{"archive_papers_tagged":3012,"with_a_code_link":622,"where_syntology_ran_a_sample":77,"not_listed_spam_title":0,"listed":3012,"listed_where_code_ran":77,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":64,"every_run_a_failure_of_syntologys_instrument":13,"listed_with_a_run_with_no_instrument_failure":64,"listed_every_run_a_failure_of_syntologys_instrument":13,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition","prev":"/task/automatic-speech-recognition","next":"/task/automatic-speech-recognition/papers/3","papers":[{"url":"/paper/from-tens-of-hours-to-tens-of-thousands","slug":"from-tens-of-hours-to-tens-of-thousands","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","date":"2025-05-22","arxiv_id":"2505.16972","repositories_listed":1,"syntology":null},{"url":"/paper/personatab-predicting-personality-traits","slug":"personatab-predicting-personality-traits","title":"PersonaTAB: Predicting Personality Traits using Textual, Acoustic, and Behavioral Cues in Fully-Duplex Speech Dialogs","date":"2025-05-20","arxiv_id":"2505.14356","repositories_listed":1,"syntology":null},{"url":"/paper/towards-inclusive-asr-investigating-voice","slug":"towards-inclusive-asr-investigating-voice","title":"Towards Inclusive ASR: Investigating Voice Conversion for Dysarthric Speech Recognition in Low-Resource Languages","date":"2025-05-20","arxiv_id":"2505.14874","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10879","slug":"2505-10879","title":"Multi-Stage Speaker Diarization for Noisy Classrooms","date":"2025-05-16","arxiv_id":"2505.10879","repositories_listed":1,"syntology":null},{"url":"/paper/vita-audio-fast-interleaved-cross-modal-token","slug":"vita-audio-fast-interleaved-cross-modal-token","title":"VITA-Audio: Fast Interleaved Cross-Modal Token Generation for Efficient Large Speech-Language Model","date":"2025-05-06","arxiv_id":"2505.03739","repositories_listed":1,"syntology":null},{"url":"/paper/voila-voice-language-foundation-models-for","slug":"voila-voice-language-foundation-models-for","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","date":"2025-05-05","arxiv_id":"2505.02707","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/voila-voice-language-foundation-models-for#ran","syntology_url":"https://syntology.ai/paper/2505.02707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02707"}},"official":{"repos":["maitrix-org/voila"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bersting-at-the-screams-a-benchmark-for","slug":"bersting-at-the-screams-a-benchmark-for","title":"BERSting at the Screams: A Benchmark for Distanced, Emotional and Shouted Speech Recognition","date":"2025-04-30","arxiv_id":"2505.00059","repositories_listed":1,"syntology":null},{"url":"/paper/docia-an-online-document-level-context","slug":"docia-an-online-document-level-context","title":"DoCIA: An Online Document-Level Context Incorporation Agent for Speech Translation","date":"2025-04-07","arxiv_id":"2504.05122","repositories_listed":1,"syntology":null},{"url":"/paper/whispering-under-the-eaves-protecting-user","slug":"whispering-under-the-eaves-protecting-user","title":"Whispering Under the Eaves: Protecting User Privacy Against Commercial and LLM-powered Automatic Speech Recognition Systems","date":"2025-04-01","arxiv_id":"2504.00858","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/whispering-under-the-eaves-protecting-user#ran","syntology_url":"https://syntology.ai/paper/2504.00858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.00858"}},"official":{"repos":["WeifeiJin/AudioShield"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dolphin-a-large-scale-automatic-speech","slug":"dolphin-a-large-scale-automatic-speech","title":"Dolphin: A Large-Scale Automatic Speech Recognition Model for Eastern Languages","date":"2025-03-26","arxiv_id":"2503.20212","repositories_listed":1,"syntology":null},{"url":"/paper/qwen2-5-omni-technical-report","slug":"qwen2-5-omni-technical-report","title":"Qwen2.5-Omni Technical Report","date":"2025-03-26","arxiv_id":"2503.20215","repositories_listed":1,"syntology":null},{"url":"/paper/cleanmel-mel-spectrogram-enhancement-for","slug":"cleanmel-mel-spectrogram-enhancement-for","title":"CleanMel: Mel-Spectrogram Enhancement for Improving Both Speech Quality and ASR","date":"2025-02-27","arxiv_id":"2502.20040","repositories_listed":1,"syntology":null},{"url":"/paper/liteasr-efficient-automatic-speech","slug":"liteasr-efficient-automatic-speech","title":"LiteASR: Efficient Automatic Speech Recognition with Low-Rank Approximation","date":"2025-02-27","arxiv_id":"2502.20583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/liteasr-efficient-automatic-speech#ran","syntology_url":"https://syntology.ai/paper/2502.20583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20583"}},"official":{"repos":["efeslab/liteasr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-inclusivity-of-dutch-speech","slug":"improving-the-inclusivity-of-dutch-speech","title":"Improving the Inclusivity of Dutch Speech Recognition by Fine-tuning Whisper on the JASMIN-CGN Corpus","date":"2025-02-24","arxiv_id":"2502.17284","repositories_listed":1,"syntology":null},{"url":"/paper/duplexmamba-enhancing-real-time-speech","slug":"duplexmamba-enhancing-real-time-speech","title":"DuplexMamba: Enhancing Real-time Speech Conversations with Duplex and Streaming Capabilities","date":"2025-02-16","arxiv_id":"2502.11123","repositories_listed":1,"syntology":null},{"url":"/paper/vinp-variational-bayesian-inference-with","slug":"vinp-variational-bayesian-inference-with","title":"VINP: Variational Bayesian Inference with Neural Speech Prior for Joint ASR-Effective Speech Dereverberation and Blind RIR Identification","date":"2025-02-11","arxiv_id":"2502.07205","repositories_listed":1,"syntology":null},{"url":"/paper/audio-visual-representation-learning-via","slug":"audio-visual-representation-learning-via","title":"Audio-Visual Representation Learning via Knowledge Distillation from Speech Foundation Models","date":"2025-02-09","arxiv_id":"2502.05766","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-broadcast-media-subtitle","slug":"leveraging-broadcast-media-subtitle","title":"Leveraging Broadcast Media Subtitle Transcripts for Automatic Speech Recognition and Subtitling","date":"2025-02-05","arxiv_id":"2502.03212","repositories_listed":1,"syntology":null},{"url":"/paper/sagalee-an-open-source-automatic-speech","slug":"sagalee-an-open-source-automatic-speech","title":"Sagalee: an Open Source Automatic Speech Recognition Dataset for Oromo Language","date":"2025-02-01","arxiv_id":"2502.00421","repositories_listed":1,"syntology":null},{"url":"/paper/speech-translation-refinement-using-large","slug":"speech-translation-refinement-using-large","title":"Speech Translation Refinement using Large Language Models","date":"2025-01-25","arxiv_id":"2501.15090","repositories_listed":1,"syntology":null},{"url":"/paper/fireredasr-open-source-industrial-grade","slug":"fireredasr-open-source-industrial-grade","title":"FireRedASR: Open-Source Industrial-Grade Mandarin Speech Recognition Models from Encoder-Decoder to LLM Integration","date":"2025-01-24","arxiv_id":"2501.14350","repositories_listed":1,"syntology":null},{"url":"/paper/flanec-exploring-flan-t5-for-post-asr-error","slug":"flanec-exploring-flan-t5-for-post-asr-error","title":"FlanEC: Exploring Flan-T5 for Post-ASR Error Correction","date":"2025-01-22","arxiv_id":"2501.12979","repositories_listed":1,"syntology":null},{"url":"/paper/selective-attention-merging-for-low-resource","slug":"selective-attention-merging-for-low-resource","title":"Selective Attention Merging for low resource tasks: A case study of Child ASR","date":"2025-01-14","arxiv_id":"2501.08468","repositories_listed":1,"syntology":null},{"url":"/paper/adacs-adaptive-normalization-for-enhanced","slug":"adacs-adaptive-normalization-for-enhanced","title":"AdaCS: Adaptive Normalization for Enhanced Code-Switching ASR","date":"2025-01-13","arxiv_id":"2501.07102","repositories_listed":1,"syntology":null},{"url":"/paper/discrete-speech-unit-extraction-via","slug":"discrete-speech-unit-extraction-via","title":"Discrete Speech Unit Extraction via Independent Component Analysis","date":"2025-01-11","arxiv_id":"2501.06562","repositories_listed":1,"syntology":null},{"url":"/paper/comparing-self-supervised-learning-models-pre","slug":"comparing-self-supervised-learning-models-pre","title":"Comparing Self-Supervised Learning Models Pre-Trained on Human Speech and Animal Vocalizations for Bioacoustics Processing","date":"2025-01-10","arxiv_id":"2501.05987","repositories_listed":1,"syntology":null},{"url":"/paper/listening-and-seeing-again-generative-error","slug":"listening-and-seeing-again-generative-error","title":"Listening and Seeing Again: Generative Error Correction for Audio-Visual Speech Recognition","date":"2025-01-03","arxiv_id":"2501.04038","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-singlish-understanding-bridging-the","slug":"advancing-singlish-understanding-bridging-the","title":"Advancing Singlish Understanding: Bridging the Gap with Datasets and Multimodal Models","date":"2025-01-02","arxiv_id":"2501.01034","repositories_listed":1,"syntology":null},{"url":"/paper/dicow-diarization-conditioned-whisper-for","slug":"dicow-diarization-conditioned-whisper-for","title":"DiCoW: Diarization-Conditioned Whisper for Target Speaker Automatic Speech Recognition","date":"2024-12-30","arxiv_id":"2501.00114","repositories_listed":1,"syntology":null},{"url":"/paper/mathspeech-leveraging-small-lms-for-accurate","slug":"mathspeech-leveraging-small-lms-for-accurate","title":"MathSpeech: Leveraging Small LMs for Accurate Conversion in Mathematical Speech-to-Formula","date":"2024-12-20","arxiv_id":"2412.15655","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-keyword-spotting-boosted-by-cross","slug":"streaming-keyword-spotting-boosted-by-cross","title":"Streaming Keyword Spotting Boosted by Cross-layer Discrimination Consistency","date":"2024-12-17","arxiv_id":"2412.12635","repositories_listed":1,"syntology":null},{"url":"/paper/glm-4-voice-towards-intelligent-and-human","slug":"glm-4-voice-towards-intelligent-and-human","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","date":"2024-12-03","arxiv_id":"2412.02612","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glm-4-voice-towards-intelligent-and-human#ran","syntology_url":"https://syntology.ai/paper/2412.02612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02612"}},"official":{"repos":["thudm/glm-4-voice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/k2ssl-a-faster-and-better-framework-for-self","slug":"k2ssl-a-faster-and-better-framework-for-self","title":"k2SSL: A Faster and Better Framework for Self-Supervised Speech Representation Learning","date":"2024-11-26","arxiv_id":"2411.17100","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-speech-text-pre-training-with","slug":"scaling-speech-text-pre-training-with","title":"Scaling Speech-Text Pre-training with Synthetic Interleaved Data","date":"2024-11-26","arxiv_id":"2411.17607","repositories_listed":1,"syntology":null},{"url":"/paper/ctc-assisted-llm-based-contextual-asr","slug":"ctc-assisted-llm-based-contextual-asr","title":"CTC-Assisted LLM-Based Contextual ASR","date":"2024-11-10","arxiv_id":"2411.06437","repositories_listed":1,"syntology":null},{"url":"/paper/dialectal-coverage-and-generalization-in","slug":"dialectal-coverage-and-generalization-in","title":"Dialectal Coverage And Generalization in Arabic Speech Recognition","date":"2024-11-07","arxiv_id":"2411.05872","repositories_listed":1,"syntology":null},{"url":"/paper/voicebench-benchmarking-llm-based-voice","slug":"voicebench-benchmarking-llm-based-voice","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","date":"2024-10-22","arxiv_id":"2410.17196","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voicebench-benchmarking-llm-based-voice#ran","syntology_url":"https://syntology.ai/paper/2410.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17196"}},"official":{"repos":["matthewcym/voicebench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-multimodal-sentiment-analysis-for","slug":"enhancing-multimodal-sentiment-analysis-for","title":"Enhancing Multimodal Sentiment Analysis for Missing Modality through Self-Distillation and Unified Modality Cross-Attention","date":"2024-10-19","arxiv_id":"2410.15029","repositories_listed":1,"syntology":null},{"url":"/paper/cr-ctc-consistency-regularization-on-ctc-for","slug":"cr-ctc-consistency-regularization-on-ctc-for","title":"CR-CTC: Consistency regularization on CTC for improved speech recognition","date":"2024-10-07","arxiv_id":"2410.05101","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-with-pre","slug":"end-to-end-speech-recognition-with-pre","title":"End-to-End Speech Recognition with Pre-trained Masked Language Model","date":"2024-10-01","arxiv_id":"2410.00528","repositories_listed":1,"syntology":null},{"url":"/paper/afrihubert-a-self-supervised-speech","slug":"afrihubert-a-self-supervised-speech","title":"AfriHuBERT: A self-supervised speech representation model for African languages","date":"2024-09-30","arxiv_id":"2409.20201","repositories_listed":1,"syntology":null},{"url":"/paper/mamba-for-streaming-asr-combined-with","slug":"mamba-for-streaming-asr-combined-with","title":"Mamba for Streaming ASR Combined with Unimodal Aggregation","date":"2024-09-30","arxiv_id":"2410.00070","repositories_listed":1,"syntology":null},{"url":"/paper/improving-multilingual-asr-in-the-wild-using","slug":"improving-multilingual-asr-in-the-wild-using","title":"Improving Multilingual ASR in the Wild Using Simple N-best Re-ranking","date":"2024-09-27","arxiv_id":"2409.18428","repositories_listed":1,"syntology":null},{"url":"/paper/weighted-cross-entropy-for-low-resource","slug":"weighted-cross-entropy-for-low-resource","title":"Weighted Cross-entropy for Low-Resource Languages in Multilingual Speech Recognition","date":"2024-09-25","arxiv_id":"2409.16954","repositories_listed":1,"syntology":null},{"url":"/paper/revise-reason-and-recognize-llm-based-emotion","slug":"revise-reason-and-recognize-llm-based-emotion","title":"Revise, Reason, and Recognize: LLM-Based Emotion Recognition via Emotion-Specific Prompts and ASR Error Correction","date":"2024-09-23","arxiv_id":"2409.15551","repositories_listed":1,"syntology":null},{"url":"/paper/2409-14074","slug":"2409-14074","title":"MultiMed: Multilingual Medical Speech Recognition via Attention Encoder Decoder","date":"2024-09-21","arxiv_id":"2409.14074","repositories_listed":1,"syntology":null},{"url":"/paper/channel-aware-domain-adaptive-generative","slug":"channel-aware-domain-adaptive-generative","title":"Channel-Aware Domain-Adaptive Generative Adversarial Network for Robust Speech Recognition","date":"2024-09-19","arxiv_id":"2409.12386","repositories_listed":1,"syntology":null},{"url":"/paper/asr-benchmarking-need-for-a-more","slug":"asr-benchmarking-need-for-a-more","title":"ASR Benchmarking: Need for a More Representative Conversational Dataset","date":"2024-09-18","arxiv_id":"2409.12042","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-strong-audio-visual","slug":"large-language-models-are-strong-audio-visual","title":"Large Language Models are Strong Audio-Visual Speech Recognition Learners","date":"2024-09-18","arxiv_id":"2409.12319","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-are-strong-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2409.12319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12319"}},"official":{"repos":["umbertocappellazzo/llama-avsr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-can-transcribe-speech-in","slug":"large-language-model-can-transcribe-speech-in","title":"Large Language Model Can Transcribe Speech in Multi-Talker Scenarios with Versatile Instructions","date":"2024-09-13","arxiv_id":"2409.08596","repositories_listed":1,"syntology":null},{"url":"/paper/linear-time-complexity-conformers-with","slug":"linear-time-complexity-conformers-with","title":"Linear Time Complexity Conformers with SummaryMixing for Streaming Speech Recognition","date":"2024-09-11","arxiv_id":"2409.07165","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-accuracy-of-automatic-speech","slug":"measuring-the-accuracy-of-automatic-speech","title":"Measuring the Accuracy of Automatic Speech Recognition Solutions","date":"2024-08-29","arxiv_id":"2408.16287","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-levenshtein-leveraging-multiple","slug":"beyond-levenshtein-leveraging-multiple","title":"Beyond Levenshtein: Leveraging Multiple Algorithms for Robust Word Error Rate Computations And Granular Error Classifications","date":"2024-08-28","arxiv_id":"2408.15616","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-speech-representations-still","slug":"self-supervised-speech-representations-still","title":"Self-supervised Speech Representations Still Struggle with African American Vernacular English","date":"2024-08-26","arxiv_id":"2408.14262","repositories_listed":1,"syntology":null},{"url":"/paper/li-tta-language-informed-test-time-adaptation","slug":"li-tta-language-informed-test-time-adaptation","title":"LI-TTA: Language Informed Test-Time Adaptation for Automatic Speech Recognition","date":"2024-08-11","arxiv_id":"2408.05769","repositories_listed":1,"syntology":null},{"url":"/paper/mooer-llm-based-speech-recognition-and","slug":"mooer-llm-based-speech-recognition-and","title":"MooER: LLM-based Speech Recognition and Translation Models from Moore Threads","date":"2024-08-09","arxiv_id":"2408.05101","repositories_listed":1,"syntology":null},{"url":"/paper/wav2graph-a-framework-for-supervised-learning","slug":"wav2graph-a-framework-for-supervised-learning","title":"wav2graph: A Framework for Supervised Learning Knowledge Graph from Speech","date":"2024-08-08","arxiv_id":"2408.04174","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01808","slug":"2408-01808","title":"ALIF: Low-Cost Adversarial Audio Attacks on Black-Box Speech Platforms using Linguistic Features","date":"2024-08-03","arxiv_id":"2408.01808","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-prompt-design-for-llm-based-post","slug":"evolutionary-prompt-design-for-llm-based-post","title":"Evolutionary Prompt Design for LLM-Based Post-ASR Error Correction","date":"2024-07-23","arxiv_id":"2407.16370","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00005","slug":"2408-00005","title":"Framework for Curating Speech Datasets and Evaluating ASR Systems: A Case Study for Polish","date":"2024-07-18","arxiv_id":"2408.00005","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-00005#ran","syntology_url":"https://syntology.ai/paper/2408.00005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00005"}},"official":{"repos":["goodmike31/pl-asr-bigos-tools"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/vibravox-a-dataset-of-french-speech-captured","slug":"vibravox-a-dataset-of-french-speech-captured","title":"Vibravox: A Dataset of French Speech Captured with Body-conduction Audio Sensors","date":"2024-07-16","arxiv_id":"2407.11828","repositories_listed":1,"syntology":null},{"url":"/paper/textless-dependency-parsing-by-labeled","slug":"textless-dependency-parsing-by-labeled","title":"Textless Dependency Parsing by Labeled Sequence Prediction","date":"2024-07-14","arxiv_id":"2407.10118","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-whisper-universal-acoustic","slug":"controlling-whisper-universal-acoustic","title":"Controlling Whisper: Universal Acoustic Adversarial Attacks to Control Speech Foundation Models","date":"2024-07-05","arxiv_id":"2407.04482","repositories_listed":1,"syntology":null},{"url":"/paper/performance-analysis-of-speech-encoders-for","slug":"performance-analysis-of-speech-encoders-for","title":"Performance Analysis of Speech Encoders for Low-Resource SLU and ASR in Tunisian Dialect","date":"2024-07-05","arxiv_id":"2407.04533","repositories_listed":1,"syntology":null},{"url":"/paper/written-term-detection-improves-spoken-term","slug":"written-term-detection-improves-spoken-term","title":"Written Term Detection Improves Spoken Term Detection","date":"2024-07-05","arxiv_id":"2407.04601","repositories_listed":1,"syntology":null},{"url":"/paper/improving-self-supervised-pre-training-using","slug":"improving-self-supervised-pre-training-using","title":"Improving Self-supervised Pre-training using Accent-Specific Codebooks","date":"2024-07-04","arxiv_id":"2407.03734","repositories_listed":1,"syntology":null},{"url":"/paper/multi-convformer-extending-conformer-with","slug":"multi-convformer-extending-conformer-with","title":"Multi-Convformer: Extending Conformer with Multiple Convolution Kernels","date":"2024-07-04","arxiv_id":"2407.03718","repositories_listed":1,"syntology":null},{"url":"/paper/pinyin-regularization-in-error-correction-for","slug":"pinyin-regularization-in-error-correction-for","title":"Pinyin Regularization in Error Correction for Chinese Speech Recognition with Large Language Models","date":"2024-07-02","arxiv_id":"2407.01909","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-asr-robustness-to-packet-loss-with-a","slug":"enhanced-asr-robustness-to-packet-loss-with-a","title":"Enhanced ASR Robustness to Packet Loss with a Front-End Adaptation Network","date":"2024-06-27","arxiv_id":"2406.18928","repositories_listed":1,"syntology":null},{"url":"/paper/arzen-llm-code-switched-egyptian-arabic","slug":"arzen-llm-code-switched-egyptian-arabic","title":"ArzEn-LLM: Code-Switched Egyptian Arabic-English Translation and Speech Recognition Using LLMs","date":"2024-06-26","arxiv_id":"2406.18120","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-for-the-nepali","slug":"automatic-speech-recognition-for-the-nepali","title":"Automatic speech recognition for the Nepali language using CNN, bidirectional LSTM and ResNet","date":"2024-06-25","arxiv_id":"2406.17825","repositories_listed":1,"syntology":null},{"url":"/paper/fasa-a-flexible-and-automatic-speech-aligner","slug":"fasa-a-flexible-and-automatic-speech-aligner","title":"FASA: a Flexible and Automatic Speech Aligner for Extracting High-quality Aligned Children Speech Data","date":"2024-06-25","arxiv_id":"2406.17926","repositories_listed":1,"syntology":null},{"url":"/paper/growing-trees-on-sounds-assessing-strategies","slug":"growing-trees-on-sounds-assessing-strategies","title":"Growing Trees on Sounds: Assessing Strategies for End-to-End Dependency Parsing of Speech","date":"2024-06-18","arxiv_id":"2406.12621","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-online-continual-learning-for","slug":"unsupervised-online-continual-learning-for","title":"Unsupervised Online Continual Learning for Automatic Speech Recognition","date":"2024-06-18","arxiv_id":"2406.12503","repositories_listed":1,"syntology":null},{"url":"/paper/continual-test-time-adaptation-for-end-to-end","slug":"continual-test-time-adaptation-for-end-to-end","title":"Continual Test-time Adaptation for End-to-end Speech Recognition on Noisy Speech","date":"2024-06-16","arxiv_id":"2406.11064","repositories_listed":1,"syntology":null},{"url":"/paper/whisper-flamingo-integrating-visual-features","slug":"whisper-flamingo-integrating-visual-features","title":"Whisper-Flamingo: Integrating Visual Features into Whisper for Audio-Visual Speech Recognition and Translation","date":"2024-06-14","arxiv_id":"2406.10082","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":8,"n_pointer_only":18,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/whisper-flamingo-integrating-visual-features#ran","syntology_url":"https://syntology.ai/paper/2406.10082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10082"}},"official":{"repos":["roudimit/whisper-flamingo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/language-complexity-and-speech-recognition","slug":"language-complexity-and-speech-recognition","title":"Language Complexity and Speech Recognition Accuracy: Orthographic Complexity Hurts, Phonological Complexity Doesn't","date":"2024-06-13","arxiv_id":"2406.09202","repositories_listed":1,"syntology":null},{"url":"/paper/laser-learning-by-aligning-self-supervised","slug":"laser-learning-by-aligning-self-supervised","title":"LASER: Learning by Aligning Self-supervised Representations of Speech for Improving Content-related Tasks","date":"2024-06-13","arxiv_id":"2406.09153","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-frame-level-ctc-alignments-using-self","slug":"guiding-frame-level-ctc-alignments-using-self","title":"Guiding Frame-Level CTC Alignments Using Self-knowledge Distillation","date":"2024-06-12","arxiv_id":"2406.07909","repositories_listed":1,"syntology":null},{"url":"/paper/speech-emotion-recognition-with-asr","slug":"speech-emotion-recognition-with-asr","title":"Speech Emotion Recognition with ASR Transcripts: A Comprehensive Study on Word Error Rate and Fusion Techniques","date":"2024-06-12","arxiv_id":"2406.08353","repositories_listed":1,"syntology":null},{"url":"/paper/towards-unsupervised-speech-recognition","slug":"towards-unsupervised-speech-recognition","title":"Towards Unsupervised Speech Recognition Without Pronunciation Models","date":"2024-06-12","arxiv_id":"2406.08380","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-unsupervised-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2406.08380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08380"}},"official":{"repos":["jeromeni/wholeword-uasr-jstti"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mhubert-147-a-compact-multilingual-hubert","slug":"mhubert-147-a-compact-multilingual-hubert","title":"mHuBERT-147: A Compact Multilingual HuBERT Model","date":"2024-06-10","arxiv_id":"2406.06371","repositories_listed":1,"syntology":null},{"url":"/paper/lipger-visually-conditioned-generative-error","slug":"lipger-visually-conditioned-generative-error","title":"LipGER: Visually-Conditioned Generative Error Correction for Robust Automatic Speech Recognition","date":"2024-06-06","arxiv_id":"2406.04432","repositories_listed":1,"syntology":null},{"url":"/paper/to-distill-or-not-to-distill-on-the","slug":"to-distill-or-not-to-distill-on-the","title":"To Distill or Not to Distill? On the Robustness of Robust Knowledge Distillation","date":"2024-06-06","arxiv_id":"2406.04512","repositories_listed":1,"syntology":null},{"url":"/paper/streamspeech-simultaneous-speech-to-speech","slug":"streamspeech-simultaneous-speech-to-speech","title":"StreamSpeech: Simultaneous Speech-to-Speech Translation with Multi-task Learning","date":"2024-06-05","arxiv_id":"2406.03049","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streamspeech-simultaneous-speech-to-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03049"}},"official":{"repos":["ictnlp/streamspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-variance-preserving-interpolation-approach","slug":"a-variance-preserving-interpolation-approach","title":"A Variance-Preserving Interpolation Approach for Diffusion Models with Applications to Single Channel Speech Enhancement and Recognition","date":"2024-05-27","arxiv_id":"2405.16952","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-and-consistency-learning-for","slug":"contrastive-and-consistency-learning-for","title":"Contrastive and Consistency Learning for Neural Noisy-Channel Model in Spoken Language Understanding","date":"2024-05-23","arxiv_id":"2405.15097","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-fuse-step-by-step-a-generative-fusion","slug":"let-s-fuse-step-by-step-a-generative-fusion","title":"Let's Fuse Step by Step: A Generative Fusion Decoding Algorithm with LLMs for Multi-modal Text Recognition","date":"2024-05-23","arxiv_id":"2405.14259","repositories_listed":1,"syntology":null},{"url":"/paper/self-taught-recognizer-toward-unsupervised","slug":"self-taught-recognizer-toward-unsupervised","title":"Self-Taught Recognizer: Toward Unsupervised Adaptation for Speech Foundation Models","date":"2024-05-23","arxiv_id":"2405.14161","repositories_listed":1,"syntology":null},{"url":"/paper/soccernet-echoes-a-soccer-game-audio","slug":"soccernet-echoes-a-soccer-game-audio","title":"SoccerNet-Echoes: A Soccer Game Audio Commentary Dataset","date":"2024-05-12","arxiv_id":"2405.07354","repositories_listed":1,"syntology":null},{"url":"/paper/muting-whisper-a-universal-acoustic","slug":"muting-whisper-a-universal-acoustic","title":"Muting Whisper: A Universal Acoustic Adversarial Attack on Speech Foundation Models","date":"2024-05-09","arxiv_id":"2405.06134","repositories_listed":1,"syntology":null},{"url":"/paper/open-implementation-and-study-of-best-rq-for","slug":"open-implementation-and-study-of-best-rq-for","title":"Open Implementation and Study of BEST-RQ for Speech Processing","date":"2024-05-07","arxiv_id":"2405.04296","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-potential-of-llm-based-asr-on","slug":"unveiling-the-potential-of-llm-based-asr-on","title":"Unveiling the Potential of LLM-Based ASR on Chinese Open-Source Datasets","date":"2024-05-03","arxiv_id":"2405.02132","repositories_listed":1,"syntology":null},{"url":"/paper/killkan-the-automatic-speech-recognition","slug":"killkan-the-automatic-speech-recognition","title":"Killkan: The Automatic Speech Recognition Dataset for Kichwa with Morphosyntactic Information","date":"2024-04-23","arxiv_id":"2404.15501","repositories_listed":1,"syntology":null},{"url":"/paper/less-peaky-and-more-accurate-ctc-forced","slug":"less-peaky-and-more-accurate-ctc-forced","title":"Less Peaky and More Accurate CTC Forced Alignment by Label Priors","date":"2024-04-22","arxiv_id":"2406.02560","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/less-peaky-and-more-accurate-ctc-forced#ran","syntology_url":"https://syntology.ai/paper/2406.02560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02560"}},"official":{"repos":["huangruizhe/audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/semantically-corrected-amharic-automatic","slug":"semantically-corrected-amharic-automatic","title":"Semantically Corrected Amharic Automatic Speech Recognition","date":"2024-04-20","arxiv_id":"2404.13362","repositories_listed":1,"syntology":null},{"url":"/paper/speechcolab-leaderboard-an-open-source","slug":"speechcolab-leaderboard-an-open-source","title":"SpeechColab Leaderboard: An Open-Source Platform for Automatic Speech Recognition Evaluation","date":"2024-03-13","arxiv_id":"2403.08196","repositories_listed":1,"syntology":null},{"url":"/paper/speech-robust-bench-a-robustness-benchmark","slug":"speech-robust-bench-a-robustness-benchmark","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","date":"2024-03-08","arxiv_id":"2403.07937","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/speech-robust-bench-a-robustness-benchmark#ran","syntology_url":"https://syntology.ai/paper/2403.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07937"}},"official":{"repos":["ahmedshah1494/speech_robust_bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/pixit-joint-training-of-speaker-diarization","slug":"pixit-joint-training-of-speaker-diarization","title":"PixIT: Joint Training of Speaker Diarization and Speech Separation from Real-world Multi-speaker Recordings","date":"2024-03-04","arxiv_id":"2403.02288","repositories_listed":1,"syntology":null},{"url":"/paper/a-cross-modal-approach-to-silent-speech-with","slug":"a-cross-modal-approach-to-silent-speech-with","title":"A Cross-Modal Approach to Silent Speech with LLM-Enhanced Recognition","date":"2024-03-02","arxiv_id":"2403.05583","repositories_listed":1,"syntology":null}],"record_sha256":"3861d51d388132cd8b910eb17ab23511108024654bf8d8bca2af76f218a2e824","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}