{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-to-text-translation/papers/2","list_of":"/task/speech-to-text-translation","task":"Speech-to-Text Translation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,146],"of":146,"counts":{"archive_papers_tagged":146,"with_a_code_link":64,"where_syntology_ran_a_sample":15,"not_listed_spam_title":0,"listed":146,"listed_where_code_ran":15,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":10,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":10,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-to-text-translation","prev":"/task/speech-to-text-translation","next":null,"papers":[{"url":null,"slug":"strategies-for-improving-low-resource-speech","title":"Strategies for improving low resource speech to text translation relying on pre-trained ASR models","date":"2023-05-31","arxiv_id":"2306.00208","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-transducer-and-attention-based-encoder","title":"Hybrid Transducer and Attention based Encoder-Decoder Modeling for Speech-to-Text Tasks","date":"2023-05-04","arxiv_id":"2305.03101","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-to-speech-translation-with","title":"Enhancing Speech-to-Speech Translation with Multiple TTS Targets","date":"2023-04-10","arxiv_id":"2304.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"google-usm-scaling-automatic-speech","title":"Google USM: Scaling Automatic Speech Recognition Beyond 100 Languages","date":"2023-03-02","arxiv_id":"2303.01037","repositories_listed":0,"syntology":null},{"url":null,"slug":"m3st-mix-at-three-levels-for-speech","title":"M3ST: Mix at Three Levels for Speech Translation","date":"2022-12-07","arxiv_id":"2212.03657","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-and-effective-unsupervised-speech-1","title":"Simple and Effective Unsupervised Speech Translation","date":"2022-10-18","arxiv_id":"2210.10191","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-alignments-improve-autoregressive","title":"CTC Alignments Improve Autoregressive Translation","date":"2022-10-11","arxiv_id":"2210.05200","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-augmented-monotonic-attention","title":"Language Model Augmented Monotonic Attention for Simultaneous Translation","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"samu-xlsr-semantically-aligned-multimodal","title":"SAMU-XLSR: Semantically-Aligned Multimodal Utterance-level Cross-Lingual Speech Representation","date":"2022-05-17","arxiv_id":"2205.08180","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-adaptive-segmentation-policy-for-end","title":"Learning Adaptive Segmentation Policy for End-to-End Simultaneous Translation","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"naist-simultaneous-speech-to-text-translation","title":"NAIST Simultaneous Speech-to-Text Translation System for IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-direct-speech-to-speech-translation","title":"Enhanced Direct Speech-to-Speech Translation Using Self-supervised Pre-training and Data Augmentation","date":"2022-04-06","arxiv_id":"2204.02967","repositories_listed":0,"syntology":null},{"url":null,"slug":"xtreme-s-evaluating-cross-lingual-speech","title":"XTREME-S: Evaluating Cross-lingual Speech Representations","date":"2022-03-21","arxiv_id":"2203.10752","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-contrastive-learning-for-speech","title":"Cross-modal Contrastive Learning for Speech Translation","date":"2021-12-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experiment-on-speech-to-text-translation","title":"An Experiment on Speech-to-Text Translation Systems for Manipuri to English on Low Resource Setting","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-sinhala-speech-recognition-through","title":"Improve Sinhala Speech Recognition Through e2e LF-MMI Model","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-attentive-regularization-to-improve","title":"Decision Attentive Regularization to Improve Simultaneous Speech Translation Systems","date":"2021-10-13","arxiv_id":"2110.15729","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-translation-by-understanding","title":"Improving Speech Translation by Understanding and Learning from the Auxiliary Text Translation Task","date":"2021-07-12","arxiv_id":"2107.05782","repositories_listed":0,"syntology":null},{"url":null,"slug":"pay-better-attention-to-attention-head","title":"Pay Better Attention to Attention: Head Selection in Multilingual and Multi-Domain Sequence Modeling","date":"2021-06-21","arxiv_id":"2106.10840","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-simultaneous-speech-to-text","title":"Direct Simultaneous Speech-to-Text Translation Assisted by Synchronized Streaming ASR","date":"2021-06-11","arxiv_id":"2106.06636","repositories_listed":0,"syntology":null},{"url":"/paper/task-aware-multi-task-learning-for-speech-to","slug":"task-aware-multi-task-learning-for-speech-to","title":"TASK AWARE MULTI-TASK LEARNING FOR SPEECH TO TEXT TASKS","date":"2021-06-10","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/towards-measuring-fairness-in-ai-the-casual","slug":"towards-measuring-fairness-in-ai-the-casual","title":"Towards Measuring Fairness in AI: the Casual Conversations Dataset","date":"2021-04-06","arxiv_id":"2104.02821","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-the-evaluation-of-simultaneous-speech","title":"Towards the evaluation of automatic simultaneous speech translation from a communicative perspective","date":"2021-03-15","arxiv_id":"2103.08364","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-modality-gap-for-speech-to-text","title":"Bridging the Modality Gap for Speech-to-Text Translation","date":"2020-10-28","arxiv_id":"2010.14920","repositories_listed":0,"syntology":null},{"url":null,"slug":"mam-masked-acoustic-modeling-for-end-to-end","title":"MAM: Masked Acoustic Modeling for End-to-End Speech-to-Text Translation","date":"2020-10-22","arxiv_id":"2010.11445","repositories_listed":0,"syntology":null},{"url":null,"slug":"subtitles-to-segmentation-improving-low-1","title":"Subtitles to Segmentation: Improving Low-Resource Speech-to-Text Translation Pipelines","date":"2020-10-19","arxiv_id":"2010.09693","repositories_listed":0,"syntology":null},{"url":"/paper/end-to-end-offline-speech-translation-system","slug":"end-to-end-offline-speech-translation-system","title":"End-to-End Offline Speech Translation System for IWSLT 2020 using Modality Agnostic Meta-Learning","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simulspeech-end-to-end-simultaneous-speech-to","title":"SimulSpeech: End-to-End Simultaneous Speech to Text Translation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-representations-improve-end","title":"Self-Supervised Representations Improve End-to-End Speech Translation","date":"2020-06-22","arxiv_id":"2006.12124","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-lingual-transfer-learning-for","title":"Improving Cross-Lingual Transfer Learning for End-to-End Speech Recognition with Speech Translation","date":"2020-06-09","arxiv_id":"2006.05474","repositories_listed":0,"syntology":null},{"url":null,"slug":"subtitles-to-segmentation-improving-low","title":"Subtitles to Segmentation: Improving Low-Resource Speech-to-TextTranslation Pipelines","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-on-end-to-end-speech-to","title":"A Comparative Study on End-to-end Speech to Text Translation","date":"2019-11-20","arxiv_id":"1911.08870","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-direct-speech-to-text","title":"Data Efficient Direct Speech-to-Text Translation with Modality Agnostic Meta-Learning","date":"2019-11-11","arxiv_id":"1911.04283","repositories_listed":0,"syntology":null},{"url":"/paper/europarl-st-a-multilingual-corpus-for-speech","slug":"europarl-st-a-multilingual-corpus-for-speech","title":"Europarl-ST: A Multilingual Corpus For Speech Translation Of Parliamentary Debates","date":"2019-11-08","arxiv_id":"1911.03167","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-asr-pretraining-for-low-resource","title":"Analyzing ASR pretraining for low-resource speech-to-text translation","date":"2019-10-23","arxiv_id":"1910.10762","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-based-model-adaptation-for-direct","title":"Instance-Based Model Adaptation For Direct Speech Translation","date":"2019-10-23","arxiv_id":"1910.10663","repositories_listed":0,"syntology":null},{"url":null,"slug":"classifying-topics-in-speech-when-all-you","title":"Cross-lingual topic prediction for speech using translations","date":"2019-08-29","arxiv_id":"1908.11425","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-transformer-for-end-to-end-speech","title":"Enhancing Transformer for End-to-end Speech-to-Text Translation","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-weakly-supervised-data-to-improve","title":"Leveraging Weakly Supervised Data to Improve End-to-End Speech-to-Text Translation","date":"2018-11-05","arxiv_id":"1811.02050","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unsupervised-speech-to-text","title":"Towards Unsupervised Speech-to-Text Translation","date":"2018-11-04","arxiv_id":"1811.01307","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-cross-modal-alignment-of-speech","title":"Unsupervised Cross-Modal Alignment of Speech and Text Embedding Spaces","date":"2018-05-18","arxiv_id":"1805.07467","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-speech-to-text-translation","title":"Low-Resource Speech-to-Text Translation","date":"2018-03-24","arxiv_id":"1803.09164","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpreting-strategies-annotation-in-the-waw","title":"Interpreting Strategies Annotation in the WAW Corpus","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-of-heterogeneous-corpora-for-training","title":"Using of heterogeneous corpora for training of an ASR system","date":"2017-06-01","arxiv_id":"1706.00321","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-speech-to-text-translation-without","title":"Towards speech-to-text translation without speech recognition","date":"2017-02-13","arxiv_id":"1702.03856","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-usfd-spoken-language-translation-system","title":"The USFD Spoken Language Translation System for IWSLT 2014","date":"2015-09-13","arxiv_id":"1509.03870","repositories_listed":0,"syntology":null}],"record_sha256":"c35b2c578c7c8871087a9617665253a1333fb93087518f5658605fb99f8f8cb3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}