{"url":"/task/speech-to-text-translation","name":"Speech-to-Text Translation","slug":"speech-to-text-translation","description_markdown":"Translate audio signals of speech in one language into text in a foreign language, either in an end-to-end or cascade manner.","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":146,"papers_with_code":64,"benchmarks":11,"benchmark_tables_in_archive":11,"benchmark_tables_shown":11,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/speech-to-text-translation-on-must-c-en-de","slug":"speech-to-text-translation-on-must-c-en-de","dataset":"MuST-C EN->DE","dataset_url":"/dataset/must-c","rows_in_archive":8,"metrics":["Case-sensitive sacreBLEU"],"first_row_in_archive_order":{"model":"Task Modulation + Multitask Learning(ASR/MT) + Data Augmentation","paper_title":"TASK AWARE MULTI-TASK LEARNING FOR SPEECH TO TEXT TASKS","paper_url":"/paper/task-aware-multi-task-learning-for-speech-to","paper_date":"2021-06-10","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/speech-to-text-translation-on-must-c-en-es","slug":"speech-to-text-translation-on-must-c-en-es","dataset":"MuST-C EN->ES","dataset_url":null,"rows_in_archive":5,"metrics":["Case-sensitive sacreBLEU"],"first_row_in_archive_order":{"model":"Transformer with Adapters","paper_title":"Lightweight Adapter Tuning for Multilingual Speech Translation","paper_url":"/paper/lightweight-adapter-tuning-for-multilingual","paper_date":"2021-06-02","arxiv_id":"2106.01463","code_links":[{"title":"formiel/fairseq","url":"https://github.com/formiel/fairseq"},{"title":"formiel/fairseq","url":"https://github.com/formiel/fairseq/blob/master/examples/speech_to_text/docs/adapters.md"}],"syntology":null}},{"leaderboard":"/sota/speech-to-text-translation-on-must-c-en-fr","slug":"speech-to-text-translation-on-must-c-en-fr","dataset":"MuST-C EN->FR","dataset_url":null,"rows_in_archive":3,"metrics":["Case-sensitive sacreBLEU"],"first_row_in_archive_order":{"model":"Dual-decoder Transformer","paper_title":"Dual-decoder Transformer for Joint Automatic Speech Recognition and Multilingual Speech Translation","paper_url":"/paper/dual-decoder-transformer-for-joint-automatic","paper_date":"2020-11-02","arxiv_id":"2011.00747","code_links":[{"title":"formiel/speech-translation","url":"https://github.com/formiel/speech-translation"}],"syntology":null}},{"leaderboard":"/sota/speech-to-text-translation-on-covost-2-eng-x","slug":"speech-to-text-translation-on-covost-2-eng-x","dataset":"CoVoST 2 eng-X","dataset_url":null,"rows_in_archive":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"SeamlessM4T Large","paper_title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","paper_url":"/paper/seamlessm4t-massively-multilingual-multimodal","paper_date":"2023-08-22","arxiv_id":"2308.11596","code_links":[{"title":"facebookresearch/seamless_communication","url":"https://github.com/facebookresearch/seamless_communication"},{"title":"facebookresearch/sonar","url":"https://github.com/facebookresearch/sonar"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-3/seamless_m4t"},{"title":"pwc-1/Paper-10","url":"https://github.com/pwc-1/Paper-10/tree/main/seamless_m4t"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/speech-to-text-translation-on-covost-2-x-eng","slug":"speech-to-text-translation-on-covost-2-x-eng","dataset":"CoVoST 2 X-eng","dataset_url":null,"rows_in_archive":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"SeamlessM4T Large","paper_title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","paper_url":"/paper/seamlessm4t-massively-multilingual-multimodal","paper_date":"2023-08-22","arxiv_id":"2308.11596","code_links":[{"title":"facebookresearch/seamless_communication","url":"https://github.com/facebookresearch/seamless_communication"},{"title":"facebookresearch/sonar","url":"https://github.com/facebookresearch/sonar"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-3/seamless_m4t"},{"title":"pwc-1/Paper-10","url":"https://github.com/pwc-1/Paper-10/tree/main/seamless_m4t"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/speech-to-text-translation-on-fleurs-eng-x","slug":"speech-to-text-translation-on-fleurs-eng-x","dataset":"FLEURS eng-X","dataset_url":null,"rows_in_archive":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"SeamlessM4T Large","paper_title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","paper_url":"/paper/seamlessm4t-massively-multilingual-multimodal","paper_date":"2023-08-22","arxiv_id":"2308.11596","code_links":[{"title":"facebookresearch/seamless_communication","url":"https://github.com/facebookresearch/seamless_communication"},{"title":"facebookresearch/sonar","url":"https://github.com/facebookresearch/sonar"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-3/seamless_m4t"},{"title":"pwc-1/Paper-10","url":"https://github.com/pwc-1/Paper-10/tree/main/seamless_m4t"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/speech-to-text-translation-on-fleurs-x-eng","slug":"speech-to-text-translation-on-fleurs-x-eng","dataset":"FLEURS X-eng","dataset_url":null,"rows_in_archive":2,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"SeamlessM4T Large","paper_title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","paper_url":"/paper/seamlessm4t-massively-multilingual-multimodal","paper_date":"2023-08-22","arxiv_id":"2308.11596","code_links":[{"title":"facebookresearch/seamless_communication","url":"https://github.com/facebookresearch/seamless_communication"},{"title":"facebookresearch/sonar","url":"https://github.com/facebookresearch/sonar"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-3/seamless_m4t"},{"title":"pwc-1/Paper-10","url":"https://github.com/pwc-1/Paper-10/tree/main/seamless_m4t"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/speech-to-text-translation-on-libri-trans","slug":"speech-to-text-translation-on-libri-trans","dataset":"libri-trans","dataset_url":null,"rows_in_archive":2,"metrics":["Case-insensitive sacreBLEU","Case-insensitive tokenized BLEU","Case-sensitive sacreBLEU","Case-sensitive tokenized BLEU"],"first_row_in_archive_order":{"model":"Transformer + ASR Pretrain + SpecAug","paper_title":"NeurST: Neural Speech Translation Toolkit","paper_url":"/paper/neurst-neural-speech-translation-toolkit","paper_date":"2020-12-18","arxiv_id":"2012.10018","code_links":[{"title":"bytedance/neurst","url":"https://github.com/bytedance/neurst"}],"syntology":null}},{"leaderboard":"/sota/speech-to-text-translation-on-medibeng","slug":"speech-to-text-translation-on-medibeng","dataset":"MediBeng","dataset_url":"/dataset/medibeng","rows_in_archive":2,"metrics":["Bleu"],"first_row_in_archive_order":{"model":"MediBeng Whisper Tiny","paper_title":"MEDIBENG WHISPER TINY: A FINE-TUNED CODE-SWITCHED BENGALI-ENGLISH TRANSLATOR FOR CLINICAL APPLICATIONS","paper_url":"/paper/medibeng-whisper-tiny-a-fine-tuned-code","paper_date":"2025-04-25","arxiv_id":null,"code_links":[{"title":"pr0mila/MediBeng-Whisper-Tiny","url":"https://github.com/pr0mila/MediBeng-Whisper-Tiny"}],"syntology":null}},{"leaderboard":"/sota/speech-to-text-translation-on-must-c-1","slug":"speech-to-text-translation-on-must-c-1","dataset":"MuST-C","dataset_url":"/dataset/must-c","rows_in_archive":2,"metrics":["SacreBLEU"],"first_row_in_archive_order":{"model":"Transformer with Adapters","paper_title":"Lightweight Adapter Tuning for Multilingual Speech Translation","paper_url":"/paper/lightweight-adapter-tuning-for-multilingual","paper_date":"2021-06-02","arxiv_id":"2106.01463","code_links":[{"title":"formiel/fairseq","url":"https://github.com/formiel/fairseq"},{"title":"formiel/fairseq","url":"https://github.com/formiel/fairseq/blob/master/examples/speech_to_text/docs/adapters.md"}],"syntology":null}},{"leaderboard":"/sota/speech-to-text-translation-on-must-c-en-nl","slug":"speech-to-text-translation-on-must-c-en-nl","dataset":"MuST-C EN->NL","dataset_url":null,"rows_in_archive":1,"metrics":["Case-sensitive sacreBLEU"],"first_row_in_archive_order":{"model":"Speechformer","paper_title":"Speechformer: Reducing Information Loss in Direct Speech Translation","paper_url":"/paper/speechformer-reducing-information-loss-in","paper_date":"2021-09-09","arxiv_id":"2109.04574","code_links":[{"title":"sarapapi/fbk-fairseq","url":"https://github.com/sarapapi/fbk-fairseq"}],"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":4}}}],"datasets":[{"url":"/dataset/must-c","name":"MuST-C","full_name":"","num_papers_in_archive":216},{"url":"/dataset/covost","name":"CoVoST","full_name":"","num_papers_in_archive":34},{"url":"/dataset/kosp2e","name":"Kosp2e","full_name":"","num_papers_in_archive":4},{"url":"/dataset/medibeng","name":"MediBeng","full_name":"Synthetic Code-Switched Bengali-English Speech Conversations for Healthcare Applications","num_papers_in_archive":1}],"subtasks":[{"url":"/task/simultaneous-speech-to-text-translation","name":"Simultaneous Speech-to-Text Translation"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":64,"tagged_in_all":146,"items":[{"url":"/paper/fairseq-s2t-fast-speech-to-text-modeling-with","title":"fairseq S2T: Fast Speech-to-Text Modeling with fairseq","date":"2020-10-11","arxiv_id":"2010.05171","repositories_listed":5,"syntology":null},{"url":"/paper/seamlessm4t-massively-multilingual-multimodal","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","date":"2023-08-22","arxiv_id":"2308.11596","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/paddlespeech-an-easy-to-use-all-in-one-speech-1","title":"PaddleSpeech: An Easy-to-Use All-in-One Speech Toolkit","date":"2022-05-20","arxiv_id":"2205.12007","repositories_listed":2,"syntology":null},{"url":"/paper/a-3-t-alignment-aware-acoustic-and-text","title":"A$^3$T: Alignment-Aware Acoustic and Text Pretraining for Speech Synthesis and Editing","date":"2022-03-18","arxiv_id":"2203.09690","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/shas-approaching-optimal-segmentation-for-end","title":"SHAS: Approaching optimal Segmentation for End-to-End Speech Translation","date":"2022-02-09","arxiv_id":"2202.04774","repositories_listed":2,"syntology":null},{"url":"/paper/lightweight-adapter-tuning-for-multilingual","title":"Lightweight Adapter Tuning for Multilingual Speech Translation","date":"2021-06-02","arxiv_id":"2106.01463","repositories_listed":2,"syntology":null},{"url":"/paper/learning-shared-semantic-space-for-speech-to","title":"Learning Shared Semantic Space for Speech-to-Text Translation","date":"2021-05-07","arxiv_id":"2105.03095","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/covost-2-a-massively-multilingual-speech-to","title":"CoVoST 2 and Massively Multilingual Speech-to-Text Translation","date":"2020-07-20","arxiv_id":"2007.10310","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/beavertalk-oregon-state-university-s-iwslt","title":"BeaverTalk: Oregon State University's IWSLT 2025 Simultaneous Speech Translation System","date":"2025-05-29","arxiv_id":"2505.24016","repositories_listed":1,"syntology":null},{"url":"/paper/audio-jailbreak-attacks-exposing","title":"Audio Jailbreak Attacks: Exposing Vulnerabilities in SpeechGPT in a White-Box Framework","date":"2025-05-24","arxiv_id":"2505.18864","repositories_listed":1,"syntology":null},{"url":"/paper/medibeng-whisper-tiny-a-fine-tuned-code","title":"MEDIBENG WHISPER TINY: A FINE-TUNED CODE-SWITCHED BENGALI-ENGLISH TRANSLATOR FOR CLINICAL APPLICATIONS","date":"2025-04-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sparqle-speech-queries-to-text-translation","title":"SparQLe: Speech Queries to Text Translation Through LLMs","date":"2025-02-13","arxiv_id":"2502.09284","repositories_listed":1,"syntology":null},{"url":"/paper/fleurs-slu-a-massively-multilingual-benchmark","title":"Fleurs-SLU: A Massively Multilingual Benchmark for Spoken Language Understanding","date":"2025-01-10","arxiv_id":"2501.06117","repositories_listed":1,"syntology":null},{"url":"/paper/llast-improved-end-to-end-speech-translation","title":"LLaST: Improved End-to-end Speech Translation System Leveraged by Large Language Models","date":"2024-07-22","arxiv_id":"2407.15415","repositories_listed":1,"syntology":null},{"url":"/paper/covoswitch-machine-translation-of-synthetic","title":"CoVoSwitch: Machine Translation of Synthetic Code-Switched Text Based on Intonation Units","date":"2024-07-19","arxiv_id":"2407.14295","repositories_listed":1,"syntology":null},{"url":"/paper/listen-and-speak-fairly-a-study-on-semantic","title":"Listen and Speak Fairly: A Study on Semantic Gender Bias in Speech Integrated Large Language Models","date":"2024-07-09","arxiv_id":"2407.06957","repositories_listed":1,"syntology":null},{"url":"/paper/voices-unheard-nlp-resources-and-models-for","title":"Voices Unheard: NLP Resources and Models for Yorùbá Regional Dialects","date":"2024-06-27","arxiv_id":"2406.19564","repositories_listed":1,"syntology":null},{"url":"/paper/arzen-llm-code-switched-egyptian-arabic","title":"ArzEn-LLM: Code-Switched Egyptian Arabic-English Translation and Speech Recognition Using LLMs","date":"2024-06-26","arxiv_id":"2406.18120","repositories_listed":1,"syntology":null},{"url":"/paper/simulseamless-fbk-at-iwslt-2024-simultaneous","title":"SimulSeamless: FBK at IWSLT 2024 Simultaneous Speech Translation","date":"2024-06-20","arxiv_id":"2406.14177","repositories_listed":1,"syntology":null},{"url":"/paper/streamatt-direct-streaming-speech-to-text","title":"StreamAtt: Direct Streaming Speech-to-Text Translation with Attention-based Audio History Selection","date":"2024-06-10","arxiv_id":"2406.06097","repositories_listed":1,"syntology":null},{"url":"/paper/streamspeech-simultaneous-speech-to-speech","title":"StreamSpeech: Simultaneous Speech-to-Speech Translation with Multi-task Learning","date":"2024-06-05","arxiv_id":"2406.03049","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/leapformer-enabling-linear-transformers-for","title":"LeaPformer: Enabling Linear Transformers for Autoregressive and Simultaneous Tasks via Learned Proportions","date":"2024-05-18","arxiv_id":"2405.13046","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/pushing-the-limits-of-zero-shot-end-to-end","title":"Pushing the Limits of Zero-shot End-to-End Speech Translation","date":"2024-02-16","arxiv_id":"2402.10422","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-zero-shot-generalizability-on","title":"Investigating Zero-Shot Generalizability on Mandarin-English Code-Switched ASR and Speech-to-text Translation of Recent Foundation Models with Self-Supervision and Weak Supervision","date":"2023-12-30","arxiv_id":"2401.00273","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-single-channel-speaker-turn-aware","title":"End-to-End Single-Channel Speaker-Turn Aware Conversational Speech Translation","date":"2023-11-01","arxiv_id":"2311.00697","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-consistency","title":"An Empirical Study of Consistency Regularization for End-to-End Speech-to-Text Translation","date":"2023-08-28","arxiv_id":"2308.14482","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-level-multimodal-and-language","title":"SONAR: Sentence-Level Multimodal and Language-Agnostic Representations","date":"2023-08-22","arxiv_id":"2308.11466","repositories_listed":1,"syntology":null},{"url":"/paper/comsl-a-composite-speech-language-model-for-1","title":"ComSL: A Composite Speech-Language Model for End-to-End Speech-to-Text Translation","date":"2023-05-24","arxiv_id":"2305.14838","repositories_listed":1,"syntology":null},{"url":"/paper/dub-discrete-unit-back-translation-for-speech","title":"DUB: Discrete Unit Back-translation for Speech Translation","date":"2023-05-19","arxiv_id":"2305.11411","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}