{"url":"/task/speech-to-text","name":"Speech-to-Text","slug":"speech-to-text","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":403,"papers_with_code":129,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/common-voice","name":"Common Voice","full_name":"Common Voice","num_papers_in_archive":449}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":129,"tagged_in_all":403,"items":[{"url":"/paper/joint-ctc-attention-based-end-to-end-speech","title":"Joint CTC-Attention based End-to-End Speech Recognition using Multi-task Learning","date":"2016-09-21","arxiv_id":"1609.06773","repositories_listed":8,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/clotho-an-audio-captioning-dataset","title":"Clotho: An Audio Captioning Dataset","date":"2019-10-21","arxiv_id":"1910.09387","repositories_listed":7,"syntology":{"n":21,"n_ran":6,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/fairseq-s2t-fast-speech-to-text-modeling-with","title":"fairseq S2T: Fast Speech-to-Text Modeling with fairseq","date":"2020-10-11","arxiv_id":"2010.05171","repositories_listed":5,"syntology":null},{"url":"/paper/seamlessm4t-massively-multilingual-multimodal","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","date":"2023-08-22","arxiv_id":"2308.11596","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/tools-and-resources-for-romanian-text-to","title":"Tools and resources for Romanian text-to-speech and speech-to-text applications","date":"2018-02-15","arxiv_id":"1802.05583","repositories_listed":4,"syntology":null},{"url":"/paper/tensor-comprehensions-framework-agnostic-high","title":"Tensor Comprehensions: Framework-Agnostic High-Performance Machine Learning Abstractions","date":"2018-02-13","arxiv_id":"1802.04730","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/audio-adversarial-examples-targeted-attacks","title":"Audio Adversarial Examples: Targeted Attacks on Speech-to-Text","date":"2018-01-05","arxiv_id":"1801.01944","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/scribosermo-fast-speech-to-text-models-for","title":"Scribosermo: Fast Speech-to-Text models for German and other Languages","date":"2021-10-15","arxiv_id":"2110.07982","repositories_listed":3,"syntology":null},{"url":"/paper/one-tts-alignment-to-rule-them-all","title":"One TTS Alignment To Rule Them All","date":"2021-08-23","arxiv_id":"2108.10447","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sequence-to","title":"Deep Reinforcement Learning For Sequence to Sequence Models","date":"2018-05-24","arxiv_id":"1805.09461","repositories_listed":3,"syntology":{"n":17,"n_ran":3,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/speechut-bridging-speech-and-text-with-hidden","title":"SpeechUT: Bridging Speech and Text with Hidden-Unit for Encoder-Decoder Based Speech-Text Pre-training","date":"2022-10-07","arxiv_id":"2210.03730","repositories_listed":2,"syntology":null},{"url":"/paper/finstreder-simple-and-fast-spoken-language","title":"Finstreder: Simple and fast Spoken Language Understanding with Finite State Transducers using modern Speech-to-Text models","date":"2022-06-29","arxiv_id":"2206.14589","repositories_listed":2,"syntology":null},{"url":"/paper/paddlespeech-an-easy-to-use-all-in-one-speech-1","title":"PaddleSpeech: An Easy-to-Use All-in-One Speech Toolkit","date":"2022-05-20","arxiv_id":"2205.12007","repositories_listed":2,"syntology":null},{"url":"/paper/a-3-t-alignment-aware-acoustic-and-text","title":"A$^3$T: Alignment-Aware Acoustic and Text Pretraining for Speech Synthesis and Editing","date":"2022-03-18","arxiv_id":"2203.09690","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/learning-shared-semantic-space-for-speech-to","title":"Learning Shared Semantic Space for Speech-to-Text Translation","date":"2021-05-07","arxiv_id":"2105.03095","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/covost-2-a-massively-multilingual-speech-to","title":"CoVoST 2 and Massively Multilingual Speech-to-Text Translation","date":"2020-07-20","arxiv_id":"2007.10310","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/common-voice-a-massively-multilingual-speech","title":"Common Voice: A Massively-Multilingual Speech Corpus","date":"2019-12-13","arxiv_id":"1912.06670","repositories_listed":2,"syntology":null},{"url":"/paper/speech-model-pre-training-for-end-to-end","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","date":"2019-04-07","arxiv_id":"1904.03670","repositories_listed":2,"syntology":null},{"url":"/paper/the-warmup-dilemma-how-learning-rate","title":"The Warmup Dilemma: How Learning Rate Strategies Impact Speech-to-Text Model Convergence","date":"2025-05-29","arxiv_id":"2505.23420","repositories_listed":1,"syntology":null},{"url":"/paper/beavertalk-oregon-state-university-s-iwslt","title":"BeaverTalk: Oregon State University's IWSLT 2025 Simultaneous Speech Translation System","date":"2025-05-29","arxiv_id":"2505.24016","repositories_listed":1,"syntology":null},{"url":"/paper/audio-jailbreak-attacks-exposing","title":"Audio Jailbreak Attacks: Exposing Vulnerabilities in SpeechGPT in a White-Box Framework","date":"2025-05-24","arxiv_id":"2505.18864","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-speech-to-speech-dialogue-modeling","title":"Enhancing Speech-to-Speech Dialogue Modeling with End-to-End Retrieval-Augmented Generation","date":"2025-04-27","arxiv_id":"2505.00028","repositories_listed":1,"syntology":null},{"url":"/paper/medibeng-whisper-tiny-a-fine-tuned-code","title":"MEDIBENG WHISPER TINY: A FINE-TUNED CODE-SWITCHED BENGALI-ENGLISH TRANSLATOR FOR CLINICAL APPLICATIONS","date":"2025-04-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-named-entity-recognition-2","title":"Transformer-Based Named Entity Recognition for Automated Server Provisioning","date":"2025-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-effect-of-transcription-noise","title":"Measuring the Effect of Transcription Noise on Downstream Language Understanding Tasks","date":"2025-02-19","arxiv_id":"2502.13645","repositories_listed":1,"syntology":null},{"url":"/paper/duplexmamba-enhancing-real-time-speech","title":"DuplexMamba: Enhancing Real-time Speech Conversations with Duplex and Streaming Capabilities","date":"2025-02-16","arxiv_id":"2502.11123","repositories_listed":1,"syntology":null},{"url":"/paper/sparqle-speech-queries-to-text-translation","title":"SparQLe: Speech Queries to Text Translation Through LLMs","date":"2025-02-13","arxiv_id":"2502.09284","repositories_listed":1,"syntology":null},{"url":"/paper/high-fidelity-simultaneous-speech-to-speech","title":"High-Fidelity Simultaneous Speech-To-Speech Translation","date":"2025-02-05","arxiv_id":"2502.03382","repositories_listed":1,"syntology":null},{"url":"/paper/osum-advancing-open-speech-understanding","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","date":"2025-01-23","arxiv_id":"2501.13306","repositories_listed":1,"syntology":null}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}