{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-to-text/papers/2","list_of":"/task/speech-to-text","task":"Speech-to-Text","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":403,"counts":{"archive_papers_tagged":403,"with_a_code_link":129,"where_syntology_ran_a_sample":24,"not_listed_spam_title":0,"listed":403,"listed_where_code_ran":24,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":18,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":18,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-to-text","prev":"/task/speech-to-text","next":"/task/speech-to-text/papers/3","papers":[{"url":"/paper/infusing-future-information-into-monotonic","slug":"infusing-future-information-into-monotonic","title":"Infusing Future Information into Monotonic Attention Through Language Models","date":"2021-09-07","arxiv_id":"2109.03121","repositories_listed":1,"syntology":null},{"url":"/paper/speech-emotion-recognition-with-multi-task","slug":"speech-emotion-recognition-with-multi-task","title":"Speech Emotion Recognition with Multi-Task Learning","date":"2021-09-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-large-scale-chinese-multimodal-ner-dataset","slug":"a-large-scale-chinese-multimodal-ner-dataset","title":"A Large-Scale Chinese Multimodal NER Dataset with Speech Clues","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/kosp2e-korean-speech-to-english-translation","slug":"kosp2e-korean-speech-to-english-translation","title":"Kosp2e: Korean Speech to English Translation Corpus","date":"2021-07-06","arxiv_id":"2107.02875","repositories_listed":1,"syntology":null},{"url":"/paper/towards-automatic-speech-to-sign-language","slug":"towards-automatic-speech-to-sign-language","title":"Towards Automatic Speech to Sign Language Generation","date":"2021-06-24","arxiv_id":"2106.12790","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-reordering-capability-in","slug":"investigating-the-reordering-capability-in","title":"Investigating the Reordering Capability in CTC-based Non-Autoregressive End-to-End Speech Translation","date":"2021-05-11","arxiv_id":"2105.04840","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-translation-via-cross-modal","slug":"end-to-end-speech-translation-via-cross-modal","title":"End-to-end Speech Translation via Cross-modal Progressive Training","date":"2021-04-21","arxiv_id":"2104.10380","repositories_listed":1,"syntology":null},{"url":"/paper/spgispeech-5000-hours-of-transcribed","slug":"spgispeech-5000-hours-of-transcribed","title":"SPGISpeech: 5,000 hours of transcribed financial audio for fully formatted end-to-end speech recognition","date":"2021-04-05","arxiv_id":"2104.02014","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-asr-system-with-automatic","slug":"end-to-end-asr-system-with-automatic","title":"End to End ASR System with Automatic Punctuation Insertion","date":"2020-12-03","arxiv_id":"2012.02012","repositories_listed":1,"syntology":null},{"url":"/paper/attentively-embracing-noise-for-robust-latent","slug":"attentively-embracing-noise-for-robust-latent","title":"Attentively Embracing Noise for Robust Latent Representation in BERT","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/de-stt-de-entaglement-of-unwanted-nuisances","slug":"de-stt-de-entaglement-of-unwanted-nuisances","title":"mask-Net: Learning Context Aware Invariant Features using Adversarial Forgetting (Student Abstract)","date":"2020-11-25","arxiv_id":"2011.12979","repositories_listed":1,"syntology":null},{"url":"/paper/iestac-english-italian-parallel-corpus-for","slug":"iestac-english-italian-parallel-corpus-for","title":"IESTAC: English-Italian Parallel Corpus for End-to-End Speech-to-Text Machine Translation","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-training-of-automatic","slug":"towards-end-to-end-training-of-automatic","title":"Towards End-to-End Training of Automatic Speech Recognition for Nigerian Pidgin","date":"2020-10-21","arxiv_id":"2010.11123","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-speech-2d-feature","slug":"end-to-end-learning-of-speech-2d-feature","title":"End-to-End Learning of Speech 2D Feature-Trajectory for Prosthetic Hands","date":"2020-09-22","arxiv_id":"2009.10283","repositories_listed":1,"syntology":null},{"url":"/paper/sdst-successive-decoding-for-speech-to-text","slug":"sdst-successive-decoding-for-speech-to-text","title":"Consecutive Decoding for Speech-to-text Translation","date":"2020-09-21","arxiv_id":"2009.09737","repositories_listed":1,"syntology":null},{"url":"/paper/ted-triple-supervision-decouples-end-to-end","slug":"ted-triple-supervision-decouples-end-to-end","title":"\"Listen, Understand and Translate\": Triple Supervision Decouples End-to-end Speech-to-text Translation","date":"2020-09-21","arxiv_id":"2009.09704","repositories_listed":1,"syntology":null},{"url":"/paper/contextualized-translation-of-automatically","slug":"contextualized-translation-of-automatically","title":"Contextualized Translation of Automatically Segmented Speech","date":"2020-08-05","arxiv_id":"2008.02270","repositories_listed":1,"syntology":null},{"url":"/paper/covost-a-diverse-multilingual-speech-to-text","slug":"covost-a-diverse-multilingual-speech-to-text","title":"CoVoST: A Diverse Multilingual Speech-To-Text Translation Corpus","date":"2020-02-04","arxiv_id":"2002.01320","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covost-a-diverse-multilingual-speech-to-text#ran","syntology_url":"https://syntology.ai/paper/2002.01320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01320"}},"official":{"repos":["facebookresearch/covost"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flexibo-cost-aware-multi-objective","slug":"flexibo-cost-aware-multi-objective","title":"FlexiBO: A Decoupled Cost-Aware Multi-Objective Optimization Approach for Deep Neural Networks","date":"2020-01-18","arxiv_id":"2001.06588","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/flexibo-cost-aware-multi-objective#ran","syntology_url":"https://syntology.ai/paper/2001.06588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.06588"}},"official":{"repos":["softsys4ai/FlexiBO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/stacked-debert-all-attention-in-incomplete","slug":"stacked-debert-all-attention-in-incomplete","title":"Stacked DeBERT: All Attention in Incomplete Data for Text Classification","date":"2020-01-01","arxiv_id":"2001.00137","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stacked-debert-all-attention-in-incomplete#ran","syntology_url":"https://syntology.ai/paper/2001.00137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00137"}},"official":{"repos":["gcunhase/StackedDeBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/synchronous-speech-recognition-and-speech-to","slug":"synchronous-speech-recognition-and-speech-to","title":"Synchronous Speech Recognition and Speech-to-Text Translation with Interactive Decoding","date":"2019-12-16","arxiv_id":"1912.07240","repositories_listed":1,"syntology":null},{"url":"/paper/re-translation-strategies-for-long-form","slug":"re-translation-strategies-for-long-form","title":"Re-Translation Strategies For Long Form, Simultaneous, Spoken Language Translation","date":"2019-12-06","arxiv_id":"1912.03393","repositories_listed":1,"syntology":null},{"url":"/paper/kurdish-sorani-speech-to-text-presenting-an","slug":"kurdish-sorani-speech-to-text-presenting-an","title":"Kurdish (Sorani) Speech to Text: Presenting an Experimental Dataset","date":"2019-11-29","arxiv_id":"1911.13087","repositories_listed":1,"syntology":null},{"url":"/paper/direct-speech-to-speech-translation-with-a","slug":"direct-speech-to-speech-translation-with-a","title":"Direct speech-to-speech translation with a sequence-to-sequence model","date":"2019-04-12","arxiv_id":"1904.06037","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-on-high-resource-speech","slug":"pre-training-on-high-resource-speech","title":"Pre-training on high-resource speech recognition improves low-resource speech-to-text translation","date":"2018-09-05","arxiv_id":"1809.01431","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-automatic-speech-translation-of","slug":"end-to-end-automatic-speech-translation-of","title":"End-to-End Automatic Speech Translation of Audiobooks","date":"2018-02-12","arxiv_id":"1802.04200","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-librispeech-with-french","slug":"augmenting-librispeech-with-french","title":"Augmenting Librispeech with French Translations: A Multimodal Corpus for Direct Speech Translation Evaluation","date":"2018-02-09","arxiv_id":"1802.03142","repositories_listed":1,"syntology":null},{"url":"/paper/listen-and-translate-a-proof-of-concept-for","slug":"listen-and-translate-a-proof-of-concept-for","title":"Listen and Translate: A Proof of Concept for End-to-End Speech-to-Text Translation","date":"2016-12-06","arxiv_id":"1612.01744","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/listen-and-translate-a-proof-of-concept-for#ran","syntology_url":"https://syntology.ai/paper/1612.01744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.01744"}},"official":{"repos":["eske/seq2seq"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-quality-assessment-for-speech","slug":"automatic-quality-assessment-for-speech","title":"Automatic Quality Assessment for Speech Translation Using Joint ASR and MT Features","date":"2016-09-20","arxiv_id":"1609.06049","repositories_listed":1,"syntology":null},{"url":null,"slug":"an-empirical-evaluation-of-ai-powered-non","title":"An Empirical Evaluation of AI-Powered Non-Player Characters' Perceived Realism and Performance in Virtual Reality Environments","date":"2025-07-14","arxiv_id":"2507.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"lm-spt-lm-aligned-semantic-distillation-for","title":"LM-SPT: LM-Aligned Semantic Distillation for Speech Tokenization","date":"2025-06-20","arxiv_id":"2506.16738","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-translation-for-low","title":"End-to-End Speech Translation for Low-Resource Languages Using Weakly Labeled Data","date":"2025-06-19","arxiv_id":"2506.16251","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-speak-and-you-find-robust-3d-visual","title":"I Speak and You Find: Robust 3D Visual Grounding with Noisy and Ambiguous Speech Inputs","date":"2025-06-17","arxiv_id":"2506.14495","repositories_listed":0,"syntology":null},{"url":null,"slug":"s2st-omni-an-efficient-and-scalable","title":"S2ST-Omni: An Efficient and Scalable Multilingual Speech-to-Speech Translation Framework via Seamless Speech-Text Alignment and Streaming Speech Generation","date":"2025-06-11","arxiv_id":"2506.11160","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08836","title":"Advancing STT for Low-Resource Real-World Speech","date":"2025-06-10","arxiv_id":"2506.08836","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-language-and-modality-transfer-in","title":"Improving Language and Modality Transfer in Translation by Character-level Modeling","date":"2025-05-30","arxiv_id":"2505.24561","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-text-translation-with-phoneme","title":"Speech-to-Text Translation with Phoneme-Augmented CoT: Enhancing Cross-Lingual Transfer in Low-Resource Scenarios","date":"2025-05-30","arxiv_id":"2505.24691","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-recommendation-system-using","title":"Conversational Recommendation System using NLP and Sentiment Analysis","date":"2025-05-17","arxiv_id":"2505.11933","repositories_listed":0,"syntology":null},{"url":null,"slug":"acquisition-of-high-quality-images-for-camera","title":"Acquisition of high-quality images for camera calibration in robotics applications via speech prompts","date":"2025-04-15","arxiv_id":"2504.11031","repositories_listed":0,"syntology":null},{"url":null,"slug":"linto-audio-and-textual-datasets-to-train-and","title":"LinTO Audio and Textual Datasets to Train and Evaluate Automatic Speech Recognition in Tunisian Arabic Dialect","date":"2025-04-03","arxiv_id":"2504.02604","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-recognition-accuracy-using","title":"Improving Speech Recognition Accuracy Using Custom Language Models with the Vosk Toolkit","date":"2025-03-26","arxiv_id":"2503.21025","repositories_listed":0,"syntology":null},{"url":null,"slug":"adast-dynamically-adapting-encoder-states-in-1","title":"AdaST: Dynamically Adapting Encoder States in the Decoder for End-to-End Speech-to-Text Translation","date":"2025-03-18","arxiv_id":"2503.14185","repositories_listed":0,"syntology":null},{"url":null,"slug":"focusing-robot-open-ended-reinforcement","title":"Focusing Robot Open-Ended Reinforcement Learning Through Users' Purposes","date":"2025-03-16","arxiv_id":"2503.12579","repositories_listed":0,"syntology":null},{"url":null,"slug":"telephone-surveys-meet-conversational-ai","title":"Telephone Surveys Meet Conversational AI: Evaluating a LLM-Based Telephone Survey System at Scale","date":"2025-02-27","arxiv_id":"2502.20140","repositories_listed":0,"syntology":null},{"url":null,"slug":"nexus-o-an-omni-perceptive-and-interactive","title":"Nexus: An Omni-Perceptive And -Interactive Model for Language, Audio, And Vision","date":"2025-02-26","arxiv_id":"2503.01879","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-speech-understanding-and-generation","title":"Balancing Speech Understanding and Generation Using Continual Pre-training for Codec-based Speech LLM","date":"2025-02-24","arxiv_id":"2502.16897","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-and-sparse-model-merging-for-multi","title":"Low-Rank and Sparse Model Merging for Multi-Lingual Speech Recognition and Translation","date":"2025-02-24","arxiv_id":"2502.17380","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-speech-translation-with","title":"Speech to Speech Translation with Translatotron: A State of the Art Review","date":"2025-02-09","arxiv_id":"2502.05980","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-end-to-end-is-overkill-rethinking","title":"When End-to-End is Overkill: Rethinking Cascaded Speech-to-Text Translation","date":"2025-02-01","arxiv_id":"2502.00377","repositories_listed":0,"syntology":null},{"url":null,"slug":"minmo-a-multimodal-large-language-model-for","title":"MinMo: A Multimodal Large Language Model for Seamless Voice Interaction","date":"2025-01-10","arxiv_id":"2501.06282","repositories_listed":0,"syntology":null},{"url":null,"slug":"existential-crisis-a-social-robot-s-reason","title":"Existential Crisis: A Social Robot's Reason for Being","date":"2025-01-06","arxiv_id":"2501.03376","repositories_listed":0,"syntology":null},{"url":null,"slug":"prepending-or-cross-attention-for-speech-to","title":"Prepending or Cross-Attention for Speech-to-Text? An Empirical Comparison","date":"2025-01-04","arxiv_id":"2501.02370","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisper-turns-stronger-augmenting-wav2vec-2-0","title":"Whisper Turns Stronger: Augmenting Wav2Vec 2.0 for Superior ASR in Low-Resource Languages","date":"2024-12-31","arxiv_id":"2501.00425","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-real-is-your-real-time-simultaneous","title":"How \"Real\" is Your Real-Time Simultaneous Speech-to-Text Translation System?","date":"2024-12-24","arxiv_id":"2412.18495","repositories_listed":0,"syntology":null},{"url":null,"slug":"greek2mathtex-a-greek-speech-to-text","title":"Greek2MathTex: A Greek Speech-to-Text Framework for LaTeX Equations Generation","date":"2024-12-11","arxiv_id":"2412.12167","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-purification-for-end-to-end","title":"Representation Purification for End-to-End Speech Translation","date":"2024-12-05","arxiv_id":"2412.04266","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-virtual-reality-and-ai-tutoring","title":"Leveraging Virtual Reality and AI Tutoring for Language Learning: A Case Study of a Virtual Campus Environment with OpenAI GPT Integration with Unity 3D","date":"2024-11-19","arxiv_id":"2411.12619","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisper-finetuning-on-nepali-language","title":"Whisper Finetuning on Nepali Language","date":"2024-11-19","arxiv_id":"2411.12587","repositories_listed":0,"syntology":null},{"url":null,"slug":"isochrony-controlled-speech-to-text","title":"Isochrony-Controlled Speech-to-Text Translation: A study on translating from Sino-Tibetan to Indo-European Languages","date":"2024-11-11","arxiv_id":"2411.07387","repositories_listed":0,"syntology":null},{"url":null,"slug":"neko-toward-post-recognition-generative","title":"NeKo: Toward Post Recognition Generative Correction Large Language Models with Task-Oriented Experts","date":"2024-11-08","arxiv_id":"2411.05945","repositories_listed":0,"syntology":null},{"url":null,"slug":"cuify-the-xr-an-open-source-package-to-embed","title":"CUIfy the XR: An Open-Source Package to Embed LLM-powered Conversational Agents in XR","date":"2024-11-07","arxiv_id":"2411.04671","repositories_listed":0,"syntology":null},{"url":null,"slug":"laser-attention-with-exponential","title":"LASER: Attention with Exponential Transformation","date":"2024-11-05","arxiv_id":"2411.03493","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-is-more-than-words-do-speech-to-text","title":"Speech is More Than Words: Do Speech-to-Text Translation Systems Leverage Prosody?","date":"2024-10-31","arxiv_id":"2410.24019","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-audio-fingerprinting","title":"Application of Audio Fingerprinting Techniques for Real-Time Scalable Speech Retrieval and Speech Clusterization","date":"2024-10-29","arxiv_id":"2410.21876","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-speech-large-language-models","title":"A Survey on Speech Large Language Models","date":"2024-10-24","arxiv_id":"2410.18908","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-biasing-to-improve-domain-specific","title":"Contextual Biasing to Improve Domain-specific Custom Vocabulary Audio Transcription without Explicit Fine-Tuning of Whisper Model","date":"2024-10-24","arxiv_id":"2410.18363","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialetto-ma-quanto-dialetto-transcribing-and","title":"Dialetto, ma Quanto Dialetto? Transcribing and Evaluating Dialects on a Continuum","date":"2024-10-18","arxiv_id":"2410.14589","repositories_listed":0,"syntology":null},{"url":null,"slug":"titanic-calling-low-bandwidth-video","title":"Titanic Calling: Low Bandwidth Video Conference from the Titanic Wreck","date":"2024-10-15","arxiv_id":"2410.11434","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-data-validation-methods-for","title":"Unsupervised Data Validation Methods for Efficient Model Training","date":"2024-10-10","arxiv_id":"2410.07880","repositories_listed":0,"syntology":null},{"url":null,"slug":"transducer-consistency-regularization-for","title":"Transducer Consistency Regularization for Speech to Text Applications","date":"2024-10-09","arxiv_id":"2410.07491","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithms-for-automatic-accentuation-and","title":"Algorithms For Automatic Accentuation And Transcription Of Russian Texts In Speech Recognition Systems","date":"2024-10-03","arxiv_id":"2410.02538","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-the-role-of-pretraining-in-direct","title":"Unveiling the Role of Pretraining in Direct Speech Translation","date":"2024-09-26","arxiv_id":"2409.18044","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-connect-speech-foundation-models-and","title":"How to Connect Speech Foundation Models and Large Language Models? What Matters and What Does Not","date":"2024-09-25","arxiv_id":"2409.17044","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-feasibility-of-fully-ai-automated","title":"On the Feasibility of Fully AI-automated Vishing Attacks","date":"2024-09-20","arxiv_id":"2409.13793","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-automated-clinical-transcriptions","title":"Toward Automated Clinical Transcriptions","date":"2024-09-20","arxiv_id":"2409.15378","repositories_listed":0,"syntology":null},{"url":null,"slug":"ideal-llm-integrating-dual-encoders-and","title":"Ideal-LLM: Integrating Dual Encoders and Language-Adapted LLM for Multilingual Speech-to-Text","date":"2024-09-17","arxiv_id":"2409.11214","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-real-time-transcriptions-using","title":"Evaluation of real-time transcriptions using end-to-end ASR models","date":"2024-09-09","arxiv_id":"2409.05674","repositories_listed":0,"syntology":null},{"url":"/paper/last-language-model-aware-speech-tokenization","slug":"last-language-model-aware-speech-tokenization","title":"LAST: Language Model Aware Speech Tokenization","date":"2024-09-05","arxiv_id":"2409.03701","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-based-ivr","title":"AI-Based IVR","date":"2024-08-20","arxiv_id":"2408.10549","repositories_listed":0,"syntology":null},{"url":null,"slug":"cmu-s-iwslt-2024-simultaneous-speech","title":"CMU's IWSLT 2024 Simultaneous Speech Translation System","date":"2024-08-14","arxiv_id":"2408.07452","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-powered-immersive-assistance-for","title":"AI-Powered Immersive Assistance for Interactive Task Execution in Industrial Environments","date":"2024-07-12","arxiv_id":"2407.09147","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-voice-command-pipelines-for-drone","title":"Evaluating Voice Command Pipelines for Drone Control: From STT and LLM to Direct Classification and Siamese Networks","date":"2024-07-10","arxiv_id":"2407.08658","repositories_listed":0,"syntology":null},{"url":null,"slug":"finetuning-end-to-end-models-for-estonian","title":"Finetuning End-to-End Models for Estonian Conversational Spoken Language Translation","date":"2024-07-04","arxiv_id":"2407.03809","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-decoder-only-large-language","title":"Investigating Decoder-only Large Language Models for Speech-to-text Translation","date":"2024-07-03","arxiv_id":"2407.03169","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unsupervised-speaker-diarization","title":"Towards Unsupervised Speaker Diarization System for Multilingual Telephone Calls Using Pre-trained Whisper Model and Mixture of Sparse Autoencoders","date":"2024-07-02","arxiv_id":"2407.01963","repositories_listed":0,"syntology":null},{"url":null,"slug":"naist-simultaneous-speech-translation-system","title":"NAIST Simultaneous Speech Translation System for IWSLT 2024","date":"2024-06-30","arxiv_id":"2407.00826","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-speech-to-text-large-language","title":"Transferable speech-to-text large language model alignment module","date":"2024-06-19","arxiv_id":"2406.13357","repositories_listed":0,"syntology":null},{"url":null,"slug":"costa-code-switched-speech-translation-using","title":"CoSTA: Code-Switched Speech Translation using Aligned Speech-Text Interleaving","date":"2024-06-16","arxiv_id":"2406.10993","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effects-of-heterogeneous-data-sources","title":"On the Effects of Heterogeneous Data Sources on Speech-to-Text Foundation Models","date":"2024-06-13","arxiv_id":"2406.09282","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-achieve-high-quality-direct-speech-to","title":"Can We Achieve High-quality Direct Speech-to-Speech Translation without Parallel Speech Data?","date":"2024-06-11","arxiv_id":"2406.07289","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-query-generation-using-large","title":"Synthetic Query Generation using Large Language Models for Virtual Assistants","date":"2024-06-10","arxiv_id":"2406.06729","repositories_listed":0,"syntology":null},{"url":null,"slug":"vr-gpt-visual-language-model-for-intelligent","title":"VR-GPT: Visual Language Model for Intelligent Virtual Reality Applications","date":"2024-05-19","arxiv_id":"2405.11537","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-mimo-systems-for-speech-to-text","title":"Semantic MIMO Systems for Speech-to-Text Transmission","date":"2024-05-13","arxiv_id":"2405.08096","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-toolchain-for-comprehensive-audio-video","title":"A Toolchain for Comprehensive Audio/Video Analysis Using Deep Learning Based Multimodal Approach (A use case of riot or violent context detection)","date":"2024-05-02","arxiv_id":"2407.03110","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-interpretation-corpus","title":"Simultaneous Interpretation Corpus Construction by Large Language Models in Distant Language Pair","date":"2024-04-18","arxiv_id":"2404.12299","repositories_listed":0,"syntology":null},{"url":null,"slug":"naturalturn-a-method-to-segment-transcripts","title":"NaturalTurn: A Method to Segment Transcripts into Naturalistic Conversational Turns","date":"2024-03-22","arxiv_id":"2403.15615","repositories_listed":0,"syntology":null},{"url":null,"slug":"rich-semantic-knowledge-enhanced-large","title":"Rich Semantic Knowledge Enhanced Large Language Models for Few-shot Chinese Spell Checking","date":"2024-03-13","arxiv_id":"2403.08492","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-semantic-communications-for-speech-to","title":"Robust Semantic Communications for Speech Transmission","date":"2024-03-08","arxiv_id":"2403.05187","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-speech-translation-models-via","title":"Compact Speech Translation Models via Discrete Speech Units Pretraining","date":"2024-02-29","arxiv_id":"2402.19333","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-punjabi-to-english-speech-translation","title":"Direct Punjabi to English speech translation using discrete units","date":"2024-02-25","arxiv_id":"2402.15967","repositories_listed":0,"syntology":null}],"record_sha256":"c5490915614337702a0881f73d9803a0db156223751a14b83e01b00db592b053","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}