{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/5","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":58,"rows_per_page":100,"rows":[401,500],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/4","next":"/task/speech-recognition-1/papers/6","papers":[{"url":"/paper/towards-unsupervised-speech-recognition","slug":"towards-unsupervised-speech-recognition","title":"Towards Unsupervised Speech Recognition Without Pronunciation Models","date":"2024-06-12","arxiv_id":"2406.08380","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-unsupervised-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2406.08380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08380"}},"official":{"repos":["jeromeni/wholeword-uasr-jstti"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/label-looping-highly-efficient-decoding-for","slug":"label-looping-highly-efficient-decoding-for","title":"Label-Looping: Highly Efficient Decoding for Transducers","date":"2024-06-10","arxiv_id":"2406.06220","repositories_listed":1,"syntology":null},{"url":"/paper/prompting-large-language-models-with-audio","slug":"prompting-large-language-models-with-audio","title":"Prompting Large Language Models with Audio for General-Purpose Speech Summarization","date":"2024-06-10","arxiv_id":"2406.05968","repositories_listed":1,"syntology":null},{"url":"/paper/do-prompts-really-prompt-exploring-the-prompt","slug":"do-prompts-really-prompt-exploring-the-prompt","title":"Do Prompts Really Prompt? Exploring the Prompt Understanding Capability of Whisper","date":"2024-06-09","arxiv_id":"2406.05806","repositories_listed":1,"syntology":null},{"url":"/paper/llm-based-speaker-diarization-correction-a","slug":"llm-based-speaker-diarization-correction-a","title":"LLM-based speaker diarization correction: A generalizable approach","date":"2024-06-07","arxiv_id":"2406.04927","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-performance-plateaus-a-comprehensive","slug":"beyond-performance-plateaus-a-comprehensive","title":"Beyond Performance Plateaus: A Comprehensive Study on Scalability in Speech Enhancement","date":"2024-06-06","arxiv_id":"2406.04269","repositories_listed":1,"syntology":null},{"url":"/paper/blsp-emo-towards-empathetic-large-speech","slug":"blsp-emo-towards-empathetic-large-speech","title":"BLSP-Emo: Towards Empathetic Large Speech-Language Models","date":"2024-06-06","arxiv_id":"2406.03872","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/blsp-emo-towards-empathetic-large-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03872"}},"official":{"repos":["cwang621/blsp-emo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/label-synchronous-neural-transducer-for-e2e","slug":"label-synchronous-neural-transducer-for-e2e","title":"Label-Synchronous Neural Transducer for E2E Simultaneous Speech Translation","date":"2024-06-06","arxiv_id":"2406.04541","repositories_listed":1,"syntology":null},{"url":"/paper/lipger-visually-conditioned-generative-error","slug":"lipger-visually-conditioned-generative-error","title":"LipGER: Visually-Conditioned Generative Error Correction for Robust Automatic Speech Recognition","date":"2024-06-06","arxiv_id":"2406.04432","repositories_listed":1,"syntology":null},{"url":"/paper/to-distill-or-not-to-distill-on-the","slug":"to-distill-or-not-to-distill-on-the","title":"To Distill or Not to Distill? On the Robustness of Robust Knowledge Distillation","date":"2024-06-06","arxiv_id":"2406.04512","repositories_listed":1,"syntology":null},{"url":"/paper/error-preserving-automatic-speech-recognition","slug":"error-preserving-automatic-speech-recognition","title":"Error-preserving Automatic Speech Recognition of Young English Learners' Language","date":"2024-06-05","arxiv_id":"2406.03235","repositories_listed":1,"syntology":null},{"url":"/paper/streamspeech-simultaneous-speech-to-speech","slug":"streamspeech-simultaneous-speech-to-speech","title":"StreamSpeech: Simultaneous Speech-to-Speech Translation with Multi-task Learning","date":"2024-06-05","arxiv_id":"2406.03049","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streamspeech-simultaneous-speech-to-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03049"}},"official":{"repos":["ictnlp/streamspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/whistle-data-efficient-multilingual-and","slug":"whistle-data-efficient-multilingual-and","title":"Whistle: Data-Efficient Multilingual and Crosslingual Speech Recognition via Weakly Phonetic Supervision","date":"2024-06-04","arxiv_id":"2406.02166","repositories_listed":1,"syntology":null},{"url":"/paper/transvip-speech-to-speech-translation-system","slug":"transvip-speech-to-speech-translation-system","title":"TransVIP: Speech to Speech Translation System with Voice and Isochrony Preservation","date":"2024-05-28","arxiv_id":"2405.17809","repositories_listed":1,"syntology":null},{"url":"/paper/a-variance-preserving-interpolation-approach","slug":"a-variance-preserving-interpolation-approach","title":"A Variance-Preserving Interpolation Approach for Diffusion Models with Applications to Single Channel Speech Enhancement and Recognition","date":"2024-05-27","arxiv_id":"2405.16952","repositories_listed":1,"syntology":null},{"url":"/paper/federating-dynamic-models-using-early-exit","slug":"federating-dynamic-models-using-early-exit","title":"Federating Dynamic Models using Early-Exit Architectures for Automatic Speech Recognition on Heterogeneous Clients","date":"2024-05-27","arxiv_id":"2405.17376","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-vision-language-action-models-for","slug":"a-survey-on-vision-language-action-models-for","title":"A Survey on Vision-Language-Action Models for Embodied AI","date":"2024-05-23","arxiv_id":"2405.14093","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-and-consistency-learning-for","slug":"contrastive-and-consistency-learning-for","title":"Contrastive and Consistency Learning for Neural Noisy-Channel Model in Spoken Language Understanding","date":"2024-05-23","arxiv_id":"2405.15097","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-fuse-step-by-step-a-generative-fusion","slug":"let-s-fuse-step-by-step-a-generative-fusion","title":"Let's Fuse Step by Step: A Generative Fusion Decoding Algorithm with LLMs for Multi-modal Text Recognition","date":"2024-05-23","arxiv_id":"2405.14259","repositories_listed":1,"syntology":null},{"url":"/paper/self-taught-recognizer-toward-unsupervised","slug":"self-taught-recognizer-toward-unsupervised","title":"Self-Taught Recognizer: Toward Unsupervised Adaptation for Speech Foundation Models","date":"2024-05-23","arxiv_id":"2405.14161","repositories_listed":1,"syntology":null},{"url":"/paper/mamba-in-speech-towards-an-alternative-to","slug":"mamba-in-speech-towards-an-alternative-to","title":"Mamba in Speech: Towards an Alternative to Self-Attention","date":"2024-05-21","arxiv_id":"2405.12609","repositories_listed":1,"syntology":null},{"url":"/paper/acoustic-modeling-for-overlapping-speech","slug":"acoustic-modeling-for-overlapping-speech","title":"Acoustic modeling for Overlapping Speech Recognition: JHU Chime-5 Challenge System","date":"2024-05-17","arxiv_id":"2405.11078","repositories_listed":1,"syntology":null},{"url":"/paper/no-more-mumbles-enhancing-robot","slug":"no-more-mumbles-enhancing-robot","title":"No More Mumbles: Enhancing Robot Intelligibility through Speech Adaptation","date":"2024-05-15","arxiv_id":"2405.09708","repositories_listed":1,"syntology":null},{"url":"/paper/rene-a-pre-trained-multi-modal-architecture","slug":"rene-a-pre-trained-multi-modal-architecture","title":"Rene: A Pre-trained Multi-modal Architecture for Auscultation of Respiratory Diseases","date":"2024-05-13","arxiv_id":"2405.07442","repositories_listed":1,"syntology":null},{"url":"/paper/soccernet-echoes-a-soccer-game-audio","slug":"soccernet-echoes-a-soccer-game-audio","title":"SoccerNet-Echoes: A Soccer Game Audio Commentary Dataset","date":"2024-05-12","arxiv_id":"2405.07354","repositories_listed":1,"syntology":null},{"url":"/paper/watch-your-mouth-silent-speech-recognition","slug":"watch-your-mouth-silent-speech-recognition","title":"Watch Your Mouth: Silent Speech Recognition with Depth Sensing","date":"2024-05-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/audio-visual-speech-recognition-based-on","slug":"audio-visual-speech-recognition-based-on","title":"Audio-Visual Speech Recognition based on Regulated Transformer and Spatio-Temporal Fusion Strategy for Driver Assistive Systems","date":"2024-05-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/muting-whisper-a-universal-acoustic","slug":"muting-whisper-a-universal-acoustic","title":"Muting Whisper: A Universal Acoustic Adversarial Attack on Speech Foundation Models","date":"2024-05-09","arxiv_id":"2405.06134","repositories_listed":1,"syntology":null},{"url":"/paper/open-implementation-and-study-of-best-rq-for","slug":"open-implementation-and-study-of-best-rq-for","title":"Open Implementation and Study of BEST-RQ for Speech Processing","date":"2024-05-07","arxiv_id":"2405.04296","repositories_listed":1,"syntology":null},{"url":"/paper/mixat-a-data-set-of-bilingual-emirati-english","slug":"mixat-a-data-set-of-bilingual-emirati-english","title":"Mixat: A Data Set of Bilingual Emirati-English Speech","date":"2024-05-04","arxiv_id":"2405.02578","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-potential-of-llm-based-asr-on","slug":"unveiling-the-potential-of-llm-based-asr-on","title":"Unveiling the Potential of LLM-Based ASR on Chinese Open-Source Datasets","date":"2024-05-03","arxiv_id":"2405.02132","repositories_listed":1,"syntology":null},{"url":"/paper/mamba-360-survey-of-state-space-models-as","slug":"mamba-360-survey-of-state-space-models-as","title":"Mamba-360: Survey of State Space Models as Transformer Alternative for Long Sequence Modelling: Methods, Applications, and Challenges","date":"2024-04-24","arxiv_id":"2404.16112","repositories_listed":1,"syntology":null},{"url":"/paper/killkan-the-automatic-speech-recognition","slug":"killkan-the-automatic-speech-recognition","title":"Killkan: The Automatic Speech Recognition Dataset for Kichwa with Morphosyntactic Information","date":"2024-04-23","arxiv_id":"2404.15501","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-neural-oscillations-during-speech","slug":"exploring-neural-oscillations-during-speech","title":"Exploring neural oscillations during speech perception via surrogate gradient spiking neural networks","date":"2024-04-22","arxiv_id":"2404.14024","repositories_listed":1,"syntology":null},{"url":"/paper/less-peaky-and-more-accurate-ctc-forced","slug":"less-peaky-and-more-accurate-ctc-forced","title":"Less Peaky and More Accurate CTC Forced Alignment by Label Priors","date":"2024-04-22","arxiv_id":"2406.02560","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/less-peaky-and-more-accurate-ctc-forced#ran","syntology_url":"https://syntology.ai/paper/2406.02560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02560"}},"official":{"repos":["huangruizhe/audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/semantically-corrected-amharic-automatic","slug":"semantically-corrected-amharic-automatic","title":"Semantically Corrected Amharic Automatic Speech Recognition","date":"2024-04-20","arxiv_id":"2404.13362","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-a-multilingual-large-language-model","slug":"teaching-a-multilingual-large-language-model","title":"Teaching a Multilingual Large Language Model to Understand Multilingual Speech via Multi-Instructional Training","date":"2024-04-16","arxiv_id":"2404.10922","repositories_listed":1,"syntology":null},{"url":"/paper/cmulab-an-open-source-framework-for-training","slug":"cmulab-an-open-source-framework-for-training","title":"CMULAB: An Open-Source Framework for Training and Deployment of Natural Language Processing Models","date":"2024-04-03","arxiv_id":"2404.02408","repositories_listed":1,"syntology":null},{"url":"/paper/braven-improving-self-supervised-pre-training","slug":"braven-improving-self-supervised-pre-training","title":"BRAVEn: Improving Self-Supervised Pre-training for Visual and Auditory Speech Recognition","date":"2024-04-02","arxiv_id":"2404.02098","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/braven-improving-self-supervised-pre-training#ran","syntology_url":"https://syntology.ai/paper/2404.02098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02098"}},"official":{"repos":["ahaliassos/raven"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/kallaama-a-transcribed-speech-dataset-about","slug":"kallaama-a-transcribed-speech-dataset-about","title":"Kallaama: A Transcribed Speech Dataset about Agriculture in the Three Most Widely Spoken Languages in Senegal","date":"2024-04-02","arxiv_id":"2404.01991","repositories_listed":1,"syntology":null},{"url":"/paper/elitr-bench-a-meeting-assistant-benchmark-for","slug":"elitr-bench-a-meeting-assistant-benchmark-for","title":"ELITR-Bench: A Meeting Assistant Benchmark for Long-Context Language Models","date":"2024-03-29","arxiv_id":"2403.20262","repositories_listed":1,"syntology":null},{"url":"/paper/phowhisper-automatic-speech-recognition-for","slug":"phowhisper-automatic-speech-recognition-for","title":"PhoWhisper: Automatic Speech Recognition for Vietnamese","date":"2024-03-27","arxiv_id":"2406.02555","repositories_listed":1,"syntology":null},{"url":"/paper/flowerformer-empowering-neural-architecture","slug":"flowerformer-empowering-neural-architecture","title":"FlowerFormer: Empowering Neural Architecture Encoding using a Flow-aware Graph Transformer","date":"2024-03-19","arxiv_id":"2403.12821","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/flowerformer-empowering-neural-architecture#ran","syntology_url":"https://syntology.ai/paper/2403.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12821"}},"official":{"repos":["y0ngjaenius/cvpr2024_flowerformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/spoken-100-a-cross-lingual-benchmarking","slug":"spoken-100-a-cross-lingual-benchmarking","title":"SpokeN-100: A Cross-Lingual Benchmarking Dataset for The Classification of Spoken Numbers in Different Languages","date":"2024-03-14","arxiv_id":"2403.09753","repositories_listed":1,"syntology":null},{"url":"/paper/speechcolab-leaderboard-an-open-source","slug":"speechcolab-leaderboard-an-open-source","title":"SpeechColab Leaderboard: An Open-Source Platform for Automatic Speech Recognition Evaluation","date":"2024-03-13","arxiv_id":"2403.08196","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-multimodal-cognitive-assistant-for","slug":"real-time-multimodal-cognitive-assistant-for","title":"Real-Time Multimodal Cognitive Assistant for Emergency Medical Services","date":"2024-03-11","arxiv_id":"2403.06734","repositories_listed":1,"syntology":null},{"url":"/paper/score-self-supervised-correspondence-fine","slug":"score-self-supervised-correspondence-fine","title":"SCORE: Self-supervised Correspondence Fine-tuning for Improved Content Representations","date":"2024-03-10","arxiv_id":"2403.06260","repositories_listed":1,"syntology":null},{"url":"/paper/speech-robust-bench-a-robustness-benchmark","slug":"speech-robust-bench-a-robustness-benchmark","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","date":"2024-03-08","arxiv_id":"2403.07937","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/speech-robust-bench-a-robustness-benchmark#ran","syntology_url":"https://syntology.ai/paper/2403.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07937"}},"official":{"repos":["ahmedshah1494/speech_robust_bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/a-study-of-dropout-induced-modality-bias-on","slug":"a-study-of-dropout-induced-modality-bias-on","title":"A Study of Dropout-Induced Modality Bias on Robustness to Missing Video Frames for Audio-Visual Speech Recognition","date":"2024-03-07","arxiv_id":"2403.04245","repositories_listed":1,"syntology":null},{"url":"/paper/language-and-speech-technology-for-central","slug":"language-and-speech-technology-for-central","title":"Language and Speech Technology for Central Kurdish Varieties","date":"2024-03-04","arxiv_id":"2403.01983","repositories_listed":1,"syntology":null},{"url":"/paper/pixit-joint-training-of-speaker-diarization","slug":"pixit-joint-training-of-speaker-diarization","title":"PixIT: Joint Training of Speaker Diarization and Speech Separation from Real-world Multi-speaker Recordings","date":"2024-03-04","arxiv_id":"2403.02288","repositories_listed":1,"syntology":null},{"url":"/paper/a-cross-modal-approach-to-silent-speech-with","slug":"a-cross-modal-approach-to-silent-speech-with","title":"A Cross-Modal Approach to Silent Speech with LLM-Enhanced Recognition","date":"2024-03-02","arxiv_id":"2403.05583","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-speech-models-for-automatic","slug":"multilingual-speech-models-for-automatic","title":"Twists, Humps, and Pebbles: Multilingual Speech Recognition Models Exhibit Gender Performance Gaps","date":"2024-02-28","arxiv_id":"2402.17954","repositories_listed":1,"syntology":null},{"url":"/paper/where-visual-speech-meets-language-vsp-llm","slug":"where-visual-speech-meets-language-vsp-llm","title":"Where Visual Speech Meets Language: VSP-LLM Framework for Efficient and Context-Aware Visual Speech Processing","date":"2024-02-23","arxiv_id":"2402.15151","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":2,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":12,"phrase":"10 ran (of which 2 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/where-visual-speech-meets-language-vsp-llm#ran","syntology_url":"https://syntology.ai/paper/2402.15151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15151"}},"official":{"repos":["sally-sh/vsp-llm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":2,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hint-high-quality-inpainting-transformer-with","slug":"hint-high-quality-inpainting-transformer-with","title":"HINT: High-quality INPainting Transformer with Mask-Aware Encoding and Enhanced Attention","date":"2024-02-22","arxiv_id":"2402.14185","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-hyenas-deal-with-human-speech-speech","slug":"how-do-hyenas-deal-with-human-speech-speech","title":"How do Hyenas deal with Human Speech? Speech Recognition and Translation with ConfHyena","date":"2024-02-20","arxiv_id":"2402.13208","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-do-hyenas-deal-with-human-speech-speech#ran","syntology_url":"https://syntology.ai/paper/2402.13208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13208"}},"official":{"repos":["hlt-mt/fbk-fairseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/owsm-ctc-an-open-encoder-only-speech","slug":"owsm-ctc-an-open-encoder-only-speech","title":"OWSM-CTC: An Open Encoder-Only Speech Foundation Model for Speech Recognition, Translation, and Language Identification","date":"2024-02-20","arxiv_id":"2402.12654","repositories_listed":1,"syntology":null},{"url":"/paper/air-bench-benchmarking-large-audio-language","slug":"air-bench-benchmarking-large-audio-language","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","date":"2024-02-12","arxiv_id":"2402.07729","repositories_listed":1,"syntology":null},{"url":"/paper/careless-whisper-speech-to-text-hallucination","slug":"careless-whisper-speech-to-text-hallucination","title":"Careless Whisper: Speech-to-Text Hallucination Harms","date":"2024-02-12","arxiv_id":"2402.08021","repositories_listed":1,"syntology":null},{"url":"/paper/deepcover-advancing-rnn-test-coverage-and","slug":"deepcover-advancing-rnn-test-coverage-and","title":"DeepCover: Advancing RNN Test Coverage and Online Error Prediction using State Machine Extraction","date":"2024-02-10","arxiv_id":"2402.06966","repositories_listed":1,"syntology":null},{"url":"/paper/it-s-never-too-late-fusing-acoustic","slug":"it-s-never-too-late-fusing-acoustic","title":"It's Never Too Late: Fusing Acoustic Information into Large Language Models for Automatic Speech Recognition","date":"2024-02-08","arxiv_id":"2402.05457","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/it-s-never-too-late-fusing-acoustic#ran","syntology_url":"https://syntology.ai/paper/2402.05457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05457"}},"official":null}},{"url":"/paper/unified-speech-text-pretraining-for-spoken","slug":"unified-speech-text-pretraining-for-spoken","title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","date":"2024-02-08","arxiv_id":"2402.05706","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":5,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unified-speech-text-pretraining-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2402.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05706"}},"official":{"repos":["naver-ai/usdm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reborn-reinforcement-learned-boundary","slug":"reborn-reinforcement-learned-boundary","title":"REBORN: Reinforcement-Learned Boundary Segmentation with Iterative Training for Unsupervised ASR","date":"2024-02-06","arxiv_id":"2402.03988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/reborn-reinforcement-learned-boundary#ran","syntology_url":"https://syntology.ai/paper/2402.03988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03988"}},"official":{"repos":["andybi7676/reborn-uasr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-sequence-transduction-through","slug":"streaming-sequence-transduction-through","title":"Streaming Sequence Transduction through Dynamic Compression","date":"2024-02-02","arxiv_id":"2402.01172","repositories_listed":1,"syntology":null},{"url":"/paper/owsm-v3-1-better-and-faster-open-whisper","slug":"owsm-v3-1-better-and-faster-open-whisper","title":"OWSM v3.1: Better and Faster Open Whisper-Style Speech Models based on E-Branchformer","date":"2024-01-30","arxiv_id":"2401.16658","repositories_listed":1,"syntology":null},{"url":"/paper/on-speaker-attribution-with-surt","slug":"on-speaker-attribution-with-surt","title":"On Speaker Attribution with SURT","date":"2024-01-28","arxiv_id":"2401.15676","repositories_listed":1,"syntology":null},{"url":"/paper/towards-event-extraction-from-speech-with","slug":"towards-event-extraction-from-speech-with","title":"Towards Event Extraction from Speech with Contextual Clues","date":"2024-01-27","arxiv_id":"2401.15385","repositories_listed":1,"syntology":null},{"url":"/paper/tdfnet-an-efficient-audio-visual-speech","slug":"tdfnet-an-efficient-audio-visual-speech","title":"TDFNet: An Efficient Audio-Visual Speech Separation Model with Top-down Fusion","date":"2024-01-25","arxiv_id":"2401.14185","repositories_listed":1,"syntology":null},{"url":"/paper/word-level-asr-quality-estimation-for","slug":"word-level-asr-quality-estimation-for","title":"Word-Level ASR Quality Estimation for Efficient Corpus Sampling and Post-Editing through Analyzing Attentions of a Reference-Free Metric","date":"2024-01-20","arxiv_id":"2401.11268","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-efficient-learners","slug":"large-language-models-are-efficient-learners","title":"Large Language Models are Efficient Learners of Noise-Robust Speech Recognition","date":"2024-01-19","arxiv_id":"2401.10446","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-efficient-learners#ran","syntology_url":"https://syntology.ai/paper/2401.10446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10446"}},"official":{"repos":["yuchen005/robustger"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multilingual-visual-speech-recognition-with-a","slug":"multilingual-visual-speech-recognition-with-a","title":"Efficient Training for Multilingual Visual Speech Recognition: Pre-training with Discretized Visual Speech Representation","date":"2024-01-18","arxiv_id":"2401.09802","repositories_listed":1,"syntology":null},{"url":"/paper/cascaded-cross-modal-transformer-for-audio","slug":"cascaded-cross-modal-transformer-for-audio","title":"Cascaded Cross-Modal Transformer for Audio-Textual Classification","date":"2024-01-15","arxiv_id":"2401.07575","repositories_listed":1,"syntology":null},{"url":"/paper/machine-perceptual-quality-evaluating-the","slug":"machine-perceptual-quality-evaluating-the","title":"Machine Perceptual Quality: Evaluating the Impact of Severe Lossy Compression on Audio and Image Models","date":"2024-01-15","arxiv_id":"2401.07957","repositories_listed":1,"syntology":null},{"url":"/paper/towards-online-sign-language-recognition-and","slug":"towards-online-sign-language-recognition-and","title":"Towards Online Continuous Sign Language Recognition and Translation","date":"2024-01-10","arxiv_id":"2401.05336","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-online-sign-language-recognition-and#ran","syntology_url":"https://syntology.ai/paper/2401.05336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05336"}},"official":{"repos":["FangyunWei/SLRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-speaker-encoding-network-for-multi","slug":"cross-speaker-encoding-network-for-multi","title":"Cross-Speaker Encoding Network for Multi-Talker Speech Recognition","date":"2024-01-08","arxiv_id":"2401.04152","repositories_listed":1,"syntology":null},{"url":"/paper/multichannel-av-wav2vec2-a-framework-for","slug":"multichannel-av-wav2vec2-a-framework-for","title":"Multichannel AV-wav2vec2: A Framework for Learning Multichannel Multi-Modal Speech Representation","date":"2024-01-07","arxiv_id":"2401.03468","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":2,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multichannel-av-wav2vec2-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2401.03468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03468"}},"official":{"repos":["zqs01/multi-channel-wav2vec2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/teles-temporal-lexeme-similarity-score-to","slug":"teles-temporal-lexeme-similarity-score-to","title":"TeLeS: Temporal Lexeme Similarity Score to Estimate Confidence in End-to-End ASR","date":"2024-01-06","arxiv_id":"2401.03251","repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-dialogue-as-a-catalyst-for-self","slug":"task-oriented-dialogue-as-a-catalyst-for-self","title":"Task Oriented Dialogue as a Catalyst for Self-Supervised Automatic Speech Recognition","date":"2024-01-04","arxiv_id":"2401.02417","repositories_listed":1,"syntology":null},{"url":"/paper/stateful-fastconformer-with-cache-based","slug":"stateful-fastconformer-with-cache-based","title":"Stateful Conformer with Cache-based Inference for Streaming Automatic Speech Recognition","date":"2023-12-27","arxiv_id":"2312.17279","repositories_listed":1,"syntology":null},{"url":"/paper/stable-distillation-regularizing-continued","slug":"stable-distillation-regularizing-continued","title":"Stable Distillation: Regularizing Continued Pre-training for Low-Resource Automatic Speech Recognition","date":"2023-12-20","arxiv_id":"2312.12783","repositories_listed":1,"syntology":null},{"url":"/paper/seq2seq-for-automatic-paraphasia-detection-in","slug":"seq2seq-for-automatic-paraphasia-detection-in","title":"Seq2seq for Automatic Paraphasia Detection in Aphasic Speech","date":"2023-12-16","arxiv_id":"2312.10518","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-computation-modules-granular","slug":"adaptive-computation-modules-granular","title":"Adaptive Computation Modules: Granular Conditional Computation For Efficient Inference","date":"2023-12-15","arxiv_id":"2312.10193","repositories_listed":1,"syntology":null},{"url":"/paper/flowmur-a-stealthy-and-practical-audio","slug":"flowmur-a-stealthy-and-practical-audio","title":"FlowMur: A Stealthy and Practical Audio Backdoor Attack with Limited Knowledge","date":"2023-12-15","arxiv_id":"2312.09665","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-autonomous-driving","slug":"large-language-models-for-autonomous-driving","title":"Personalized Autonomous Driving with Large Language Models: Field Experiments","date":"2023-12-14","arxiv_id":"2312.09397","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-for-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/2312.09397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09397"}},"official":null}},{"url":"/paper/extending-whisper-with-prompt-tuning-to","slug":"extending-whisper-with-prompt-tuning-to","title":"Extending Whisper with prompt tuning to target-speaker ASR","date":"2023-12-13","arxiv_id":"2312.08079","repositories_listed":1,"syntology":null},{"url":"/paper/rose-a-recognition-oriented-speech","slug":"rose-a-recognition-oriented-speech","title":"ROSE: A Recognition-Oriented Speech Enhancement Framework in Air Traffic Control Using Multi-Objective Learning","date":"2023-12-11","arxiv_id":"2312.06118","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolutions-enrich-the-self-attention","slug":"graph-convolutions-enrich-the-self-attention","title":"Graph Convolutions Enrich the Self-Attention in Transformers!","date":"2023-12-07","arxiv_id":"2312.04234","repositories_listed":1,"syntology":{"n":29,"n_ran":22,"n_constructed":0,"n_ran_checked":21,"n_instrument":1,"n_unverified":7,"n_honours":2,"n_violates":2,"n_no_contract":17,"n_pointer_only":10,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 2 honoured, 2 violated, 17 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/graph-convolutions-enrich-the-self-attention#ran","syntology_url":"https://syntology.ai/paper/2312.04234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04234"}},"official":{"repos":["jeongwhanchoi/gfsa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/bigger-is-not-always-better-the-effect-of","slug":"bigger-is-not-always-better-the-effect-of","title":"Bigger is not Always Better: The Effect of Context Size on Speech Pre-Training","date":"2023-12-03","arxiv_id":"2312.01515","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-cache-to-enable-slu-on-tiny","slug":"leveraging-cache-to-enable-slu-on-tiny","title":"Speech Understanding on Tiny Devices with A Learning Cache","date":"2023-11-30","arxiv_id":"2311.18188","repositories_listed":1,"syntology":null},{"url":"/paper/d4am-a-general-denoising-framework-for","slug":"d4am-a-general-denoising-framework-for","title":"D4AM: A General Denoising Framework for Downstream Acoustic Models","date":"2023-11-28","arxiv_id":"2311.16595","repositories_listed":1,"syntology":null},{"url":"/paper/a-quantitative-approach-to-understand-self","slug":"a-quantitative-approach-to-understand-self","title":"A Quantitative Approach to Understand Self-Supervised Models as Cross-lingual Feature Extractors","date":"2023-11-27","arxiv_id":"2311.15954","repositories_listed":1,"syntology":null},{"url":"/paper/do-vsr-models-generalize-beyond-lrs3","slug":"do-vsr-models-generalize-beyond-lrs3","title":"Do VSR Models Generalize Beyond LRS3?","date":"2023-11-23","arxiv_id":"2311.14063","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-weight-perturbed-deep-neural","slug":"investigating-weight-perturbed-deep-neural","title":"Investigating Weight-Perturbed Deep Neural Networks With Application in Iris Presentation Attack Detection","date":"2023-11-21","arxiv_id":"2311.12764","repositories_listed":1,"syntology":null},{"url":"/paper/lip-rtve-an-audiovisual-database-for-1","slug":"lip-rtve-an-audiovisual-database-for-1","title":"LIP-RTVE: An Audiovisual Database for Continuous Spanish in the Wild","date":"2023-11-21","arxiv_id":"2311.12457","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-emergent-audio","slug":"investigating-the-emergent-audio","title":"Investigating the Emergent Audio Classification Ability of ASR Foundation Models","date":"2023-11-15","arxiv_id":"2311.09363","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-audio-captioning-with-audio","slug":"zero-shot-audio-captioning-with-audio","title":"Zero-shot audio captioning with audio-language model guidance and audio context keywords","date":"2023-11-14","arxiv_id":"2311.08396","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/zero-shot-audio-captioning-with-audio#ran","syntology_url":"https://syntology.ai/paper/2311.08396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08396"}},"official":{"repos":["explainableml/zeraucap"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-in-the-context-of-precision","slug":"chatgpt-in-the-context-of-precision","title":"ChatGPT in the context of precision agriculture data analytics","date":"2023-11-10","arxiv_id":"2311.06390","repositories_listed":1,"syntology":null},{"url":"/paper/improving-whispered-speech-recognition","slug":"improving-whispered-speech-recognition","title":"Improving Whispered Speech Recognition Performance using Pseudo-whispered based Data Augmentation","date":"2023-11-09","arxiv_id":"2311.05179","repositories_listed":1,"syntology":null},{"url":"/paper/gpu-accelerated-wfst-beam-search-decoder-for","slug":"gpu-accelerated-wfst-beam-search-decoder-for","title":"GPU-Accelerated WFST Beam Search Decoder for CTC-based Speech Recognition","date":"2023-11-08","arxiv_id":"2311.04996","repositories_listed":1,"syntology":null},{"url":"/paper/improved-child-text-to-speech-synthesis","slug":"improved-child-text-to-speech-synthesis","title":"Improved Child Text-to-Speech Synthesis through Fastpitch-based Transfer Learning","date":"2023-11-07","arxiv_id":"2311.04313","repositories_listed":1,"syntology":null}],"record_sha256":"012b5016f33eb32be5a86de15d76610aa4271b31be874a4f96ec8bfceafa8179","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}