{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech/papers/5","list_of":"/task/text-to-speech","task":"Text to Speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":15,"rows_per_page":100,"rows":[401,500],"of":1419,"counts":{"archive_papers_tagged":1419,"with_a_code_link":399,"where_syntology_ran_a_sample":108,"not_listed_spam_title":0,"listed":1419,"listed_where_code_ran":108,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech","prev":"/task/text-to-speech/papers/4","next":"/task/text-to-speech/papers/6","papers":[{"url":null,"slug":"nonverbaltts-a-public-english-corpus-of-text","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","date":"2025-07-17","arxiv_id":"2507.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"p-808-multilingual-speech-enhancement-testing","title":"P.808 Multilingual Speech Enhancement Testing: Approach and Results of URGENT 2025 Challenge","date":"2025-07-15","arxiv_id":"2507.11306","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-evaluation-of-ai-powered-non","title":"An Empirical Evaluation of AI-Powered Non-Player Characters' Perceived Realism and Performance in Virtual Reality Environments","date":"2025-07-14","arxiv_id":"2507.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-leaderboards-for-large-scale","title":"Exploiting Leaderboards for Large-Scale Distribution of Malicious Models","date":"2025-07-11","arxiv_id":"2507.08983","repositories_listed":0,"syntology":null},{"url":null,"slug":"midi-valle-improving-expressive-piano","title":"MIDI-VALLE: Improving Expressive Piano Performance Synthesis Through Neural Codec Language Modelling","date":"2025-07-11","arxiv_id":"2507.08530","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-quality-assessment-model-based-on","title":"Speech Quality Assessment Model Based on Mixture of Experts: System-Level Performance Enhancement and Utterance-Level Challenge Analysis","date":"2025-07-08","arxiv_id":"2507.06116","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exploration-of-ecapa-tdnn-and-x-vector","title":"An Exploration of ECAPA-TDNN and x-vector Speaker Representations in Zero-shot Multi-speaker TTS","date":"2025-06-25","arxiv_id":"2506.20190","repositories_listed":0,"syntology":null},{"url":null,"slug":"ttsds2-resources-and-benchmark-for-evaluating","title":"TTSDS2: Resources and Benchmark for Evaluating Human-Quality Text to Speech Systems","date":"2025-06-24","arxiv_id":"2506.19441","repositories_listed":0,"syntology":null},{"url":null,"slug":"lm-spt-lm-aligned-semantic-distillation-for","title":"LM-SPT: LM-Aligned Semantic Distillation for Speech Tokenization","date":"2025-06-20","arxiv_id":"2506.16738","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-multilingual-text-to-speech-with","title":"Optimizing Multilingual Text-To-Speech with Accents & Emotions","date":"2025-06-19","arxiv_id":"2506.16310","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-non-autoregressive-model-for-accent","title":"Streaming Non-Autoregressive Model for Accent Conversion and Pronunciation Improvement","date":"2025-06-19","arxiv_id":"2506.16580","repositories_listed":0,"syntology":null},{"url":null,"slug":"predgen-accelerated-inference-of-large","title":"PredGen: Accelerated Inference of Large Language Models through Input-Time Speculation for Real-Time Speech Interaction","date":"2025-06-18","arxiv_id":"2506.15556","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonikud-hebrew-grapheme-to-phoneme","title":"Phonikud: Hebrew Grapheme-to-Phoneme Conversion for Real-Time Text-to-Speech","date":"2025-06-14","arxiv_id":"2506.12311","repositories_listed":0,"syntology":null},{"url":null,"slug":"streammel-real-time-zero-shot-text-to-speech","title":"StreamMel: Real-Time Zero-shot Text-to-Speech via Interleaved Continuous Autoregressive Modeling","date":"2025-06-14","arxiv_id":"2506.12570","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10299","title":"Scheduled Interleaved Speech-Text Training for Speech-to-Speech Translation with LLMs","date":"2025-06-12","arxiv_id":"2506.10299","repositories_listed":0,"syntology":null},{"url":null,"slug":"s2st-omni-an-efficient-and-scalable","title":"S2ST-Omni: An Efficient and Scalable Multilingual Speech-to-Speech Translation Framework via Seamless Speech-Text Alignment and Streaming Speech Generation","date":"2025-06-11","arxiv_id":"2506.11160","repositories_listed":0,"syntology":null},{"url":null,"slug":"umbratts-adapting-text-to-speech-to","title":"UmbraTTS: Adapting Text-to-Speech to Environmental Contexts with Flow Matching","date":"2025-06-11","arxiv_id":"2506.09874","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-self-refining-framework-for-enhancing-asr","title":"A Self-Refining Framework for Enhancing ASR Using TTS-Synthesized Data","date":"2025-06-10","arxiv_id":"2506.11130","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-voices-generating-a-roll-video-from","title":"Seeing Voices: Generating A-Roll Video from Audio with Mirage","date":"2025-06-09","arxiv_id":"2506.08279","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcript-prompted-whisper-with-dictionary","title":"Transcript-Prompted Whisper with Dictionary-Enhanced Decoding for Japanese Speech Annotation","date":"2025-06-09","arxiv_id":"2506.07646","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-05688","title":"Voice Impression Control in Zero-Shot TTS","date":"2025-06-06","arxiv_id":"2506.05688","repositories_listed":0,"syntology":null},{"url":null,"slug":"grapheme-coherent-phonemic-and-prosodic","title":"Grapheme-Coherent Phonemic and Prosodic Annotation of Speech by Implicit and Explicit Grapheme Conditioning","date":"2025-06-05","arxiv_id":"2506.04527","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligibility-of-text-to-speech-systems-for","title":"Intelligibility of Text-to-Speech Systems for Mathematical Expressions","date":"2025-06-05","arxiv_id":"2506.11086","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-data-augmentation-approach-for","title":"A Novel Data Augmentation Approach for Automatic Speaking Assessment on Opinion Expressions","date":"2025-06-04","arxiv_id":"2506.04077","repositories_listed":0,"syntology":null},{"url":null,"slug":"bittts-highly-compact-text-to-speech-using-1","title":"BitTTS: Highly Compact Text-to-Speech Using 1.58-bit Quantization and Weight Indexing","date":"2025-06-04","arxiv_id":"2506.03515","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-reconstruct-a-dysarthric-voice-with","title":"Can we reconstruct a dysarthric voice with the large speech model Parler TTS?","date":"2025-06-04","arxiv_id":"2506.04397","repositories_listed":0,"syntology":null},{"url":null,"slug":"hifitts-2-a-large-scale-high-bandwidth-speech","title":"HiFiTTS-2: A Large-Scale High Bandwidth Speech Dataset","date":"2025-06-04","arxiv_id":"2506.04152","repositories_listed":0,"syntology":null},{"url":null,"slug":"unicue-unified-recognition-and-generation","title":"UniCUE: Unified Recognition and Generation Framework for Chinese Cued Speech Video-to-Speech Generation","date":"2025-06-04","arxiv_id":"2506.04134","repositories_listed":0,"syntology":null},{"url":null,"slug":"capspeech-enabling-downstream-applications-in","title":"CapSpeech: Enabling Downstream Applications in Style-Captioned Text-to-Speech","date":"2025-06-03","arxiv_id":"2506.02863","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-unseen-emotion-zero-shot-expressive","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","date":"2025-06-03","arxiv_id":"2506.02742","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-japanese-full-duplex-spoken","title":"Towards a Japanese Full-duplex Spoken Dialogue System","date":"2025-06-03","arxiv_id":"2506.02979","repositories_listed":0,"syntology":null},{"url":null,"slug":"salf-mos-speaker-agnostic-latent-features","title":"SALF-MOS: Speaker Agnostic Latent Features Downsampled for MOS Prediction","date":"2025-06-02","arxiv_id":"2506.02082","repositories_listed":0,"syntology":null},{"url":null,"slug":"wctc-biasing-retraining-free-contextual","title":"WCTC-Biasing: Retraining-free Contextual Biasing ASR with Wildcard CTC-based Keyword Spotting and Inter-layer Biasing","date":"2025-06-02","arxiv_id":"2506.01263","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-for-vietnamese","title":"Zero-Shot Text-to-Speech for Vietnamese","date":"2025-06-02","arxiv_id":"2506.01322","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-activation-editing-for-post","title":"Counterfactual Activation Editing for Post-hoc Prosody and Mispronunciation Correction in TTS Models","date":"2025-06-01","arxiv_id":"2506.00832","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-thought-training-for-open-e2e-spoken","title":"Chain-of-Thought Training for Open E2E Spoken Dialogue Systems","date":"2025-05-31","arxiv_id":"2506.00722","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-token-prediction-via-compressed-to","title":"Speech Token Prediction via Compressed-to-fine Language Modeling for Speech Generation","date":"2025-05-30","arxiv_id":"2505.24496","repositories_listed":0,"syntology":null},{"url":null,"slug":"werewolf-a-straightforward-game-framework","title":"Werewolf: A Straightforward Game Framework with TTS for Improved User Engagement","date":"2025-05-30","arxiv_id":"2506.00160","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-emotion-fool-anti-spoofing","title":"Can Emotion Fool Anti-spoofing?","date":"2025-05-29","arxiv_id":"2505.23962","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-synth4kws-scalable-automatic-generation","title":"LLM-Synth4KWS: Scalable Automatic Generation and Synthesis of Confusable Data for Custom Keyword Spotting","date":"2025-05-29","arxiv_id":"2505.22995","repositories_listed":0,"syntology":null},{"url":null,"slug":"spotlight-tts-spotlighting-the-style-via","title":"Spotlight-TTS: Spotlighting the Style via Voiced-Aware Style Extraction and Style Direction Adjustment for Expressive Text-to-Speech","date":"2025-05-27","arxiv_id":"2505.20868","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-flow-matching-based-text-to","title":"Accelerating Flow-Matching-Based Text-to-Speech via Empirically Pruned Step Sampling","date":"2025-05-26","arxiv_id":"2505.19931","repositories_listed":0,"syntology":null},{"url":null,"slug":"diemo-tts-disentangled-emotion","title":"DiEmo-TTS: Disentangled Emotion Representations via Self-Supervised Distillation for Cross-Speaker Emotion Transfer in Text-to-Speech","date":"2025-05-26","arxiv_id":"2505.19687","repositories_listed":0,"syntology":null},{"url":null,"slug":"kit-s-low-resource-speech-translation-systems","title":"KIT's Low-resource Speech Translation Systems for IWSLT2025: System Enhancement with Synthetic Data and Model Regularization","date":"2025-05-26","arxiv_id":"2505.19679","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-streaming-text-to-speech-synthesis","title":"Zero-Shot Streaming Text to Speech Synthesis with Transducer and Auto-Regressive Modeling","date":"2025-05-26","arxiv_id":"2505.19669","repositories_listed":0,"syntology":null},{"url":null,"slug":"cloneshield-a-framework-for-universal","title":"CloneShield: A Framework for Universal Perturbation Against Zero-Shot Voice Cloning","date":"2025-05-25","arxiv_id":"2505.19119","repositories_listed":0,"syntology":null},{"url":null,"slug":"revival-with-voice-multi-modal-controllable","title":"Revival with Voice: Multi-modal Controllable Text-to-Speech Synthesis","date":"2025-05-25","arxiv_id":"2505.18972","repositories_listed":0,"syntology":null},{"url":null,"slug":"speakstream-streaming-text-to-speech-with","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","date":"2025-05-25","arxiv_id":"2505.19206","repositories_listed":0,"syntology":null},{"url":null,"slug":"mpe-tts-customized-emotion-zero-shot-text-to","title":"MPE-TTS: Customized Emotion Zero-Shot Text-To-Speech Using Multi-Modal Prompt","date":"2025-05-24","arxiv_id":"2505.18453","repositories_listed":0,"syntology":null},{"url":null,"slug":"rasmalai-resources-for-adaptive-speech","title":"RASMALAI: Resources for Adaptive Speech Modeling in Indian Languages with Accents and Intonations","date":"2025-05-24","arxiv_id":"2505.18609","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-you-read-isn-t-what-you-hear-linguistic","title":"What You Read Isn't What You Hear: Linguistic Sensitivity in Deepfake Speech Detection","date":"2025-05-23","arxiv_id":"2505.17513","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-expressive-japanese-character","title":"Benchmarking Expressive Japanese Character Text-to-Speech with VITS and Style-BERT-VITS2","date":"2025-05-22","arxiv_id":"2505.17320","repositories_listed":0,"syntology":null},{"url":null,"slug":"miku-pal-an-automated-and-standardized-multi","title":"MIKU-PAL: An Automated and Standardized Multi-Modal Method for Speech Paralinguistic and Affect Labeling","date":"2025-05-21","arxiv_id":"2505.15772","repositories_listed":0,"syntology":null},{"url":null,"slug":"segmentation-variant-codebooks-for","title":"Segmentation-Variant Codebooks for Preservation of Paralinguistic and Prosodic Information","date":"2025-05-21","arxiv_id":"2505.15667","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicing-personas-rewriting-persona","title":"Voicing Personas: Rewriting Persona Descriptions into Style Prompts for Controllable Text-to-Speech","date":"2025-05-21","arxiv_id":"2505.17093","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiojailbreak-jailbreak-attacks-against-end","title":"AudioJailbreak: Jailbreak Attacks against End-to-End Large Audio-Language Models","date":"2025-05-20","arxiv_id":"2505.14103","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmsd-tts-few-shot-multi-speaker-multi-dialect","title":"FMSD-TTS: Few-shot Multi-Speaker Multi-Dialect Text-to-Speech Synthesis for Ü-Tsang, Amdo and Kham Speech Dataset Generation","date":"2025-05-20","arxiv_id":"2505.14351","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-frame-rates-on-speech-tokenizer-a","title":"Impact of Frame Rates on Speech Tokenizer: A Case Study on Mandarin and English","date":"2025-05-20","arxiv_id":"2505.17076","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-noise-robustness-of-llm-based-zero","title":"Improving Noise Robustness of LLM-based Zero-shot TTS via Discrete Acoustic Token Denoising","date":"2025-05-20","arxiv_id":"2505.13830","repositories_listed":0,"syntology":null},{"url":null,"slug":"seamlessedit-background-noise-aware-zero-shot","title":"SeamlessEdit: Background Noise Aware Zero-Shot Speech Editing with in-Context Enhancement","date":"2025-05-20","arxiv_id":"2505.14066","repositories_listed":0,"syntology":null},{"url":null,"slug":"ozspeech-one-step-zero-shot-speech-synthesis","title":"OZSpeech: One-step Zero-shot Speech Synthesis with Learned-Prior-Conditioned Flow Matching","date":"2025-05-19","arxiv_id":"2505.12800","repositories_listed":0,"syntology":null},{"url":null,"slug":"shallow-flow-matching-for-coarse-to-fine-text","title":"Shallow Flow Matching for Coarse-to-Fine Text-to-Speech Synthesis","date":"2025-05-18","arxiv_id":"2505.12226","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11200","title":"Audio Turing Test: Benchmarking the Human-likeness of Large Language Model-based Text-to-Speech Systems in Chinese","date":"2025-05-16","arxiv_id":"2505.11200","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10599","title":"UDDETTS: Unifying Discrete and Dimensional Emotions for Controllable Emotional Text-to-Speech","date":"2025-05-15","arxiv_id":"2505.10599","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-end-to-end-text-to-speech","title":"Lightweight End-to-end Text-to-speech Synthesis for low resource on-device applications","date":"2025-05-12","arxiv_id":"2505.07701","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimax-speech-intrinsic-zero-shot-text-to","title":"MiniMax-Speech: Intrinsic Zero-Shot Text-to-Speech with a Learnable Speaker Encoder","date":"2025-05-12","arxiv_id":"2505.07916","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-an-intermediate-language-for","title":"Bridging the Gap: An Intermediate Language for Enhanced and Cost-Effective Grapheme-to-Phoneme Conversion with Homographs with Multiple Pronunciations Disambiguation","date":"2025-05-10","arxiv_id":"2505.06599","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexspeech-towards-stable-controllable-and","title":"FlexSpeech: Towards Stable, Controllable and Expressive Text-to-Speech","date":"2025-05-08","arxiv_id":"2505.05159","repositories_listed":0,"syntology":null},{"url":null,"slug":"teochew-wild-the-first-in-the-wild-teochew","title":"Teochew-Wild: The First In-the-wild Teochew Dataset with Orthographic Annotations","date":"2025-05-08","arxiv_id":"2505.05056","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-narrated-lecture-videos-from","title":"Generating Narrated Lecture Videos from Slides with Synchronized Highlights","date":"2025-05-05","arxiv_id":"2505.02966","repositories_listed":0,"syntology":null},{"url":null,"slug":"sadeed-advancing-arabic-diacritization","title":"Sadeed: Advancing Arabic Diacritization Through Small Language Model","date":"2025-04-30","arxiv_id":"2504.21635","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-flow-matching-based-tts-without","title":"Towards Flow-Matching-based TTS without Classifier-Free Guidance","date":"2025-04-29","arxiv_id":"2504.20334","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-framework-for-automated-1","title":"A Multi-Agent Framework for Automated Qinqiang Opera Script Generation Using Large Language Models","date":"2025-04-22","arxiv_id":"2504.15552","repositories_listed":0,"syntology":null},{"url":null,"slug":"emovoice-llm-based-emotional-text-to-speech","title":"EmoVoice: LLM-based Emotional Text-To-Speech Model with Freestyle Text Prompting","date":"2025-04-17","arxiv_id":"2504.12867","repositories_listed":0,"syntology":null},{"url":null,"slug":"goat-tts-llm-based-text-to-speech-generation","title":"GOAT-TTS: Expressive and Realistic Speech Generation via A Dual-Branch LLM","date":"2025-04-15","arxiv_id":"2504.12339","repositories_listed":0,"syntology":null},{"url":null,"slug":"autostyle-tts-retrieval-augmented-generation","title":"AutoStyle-TTS: Retrieval-Augmented Generation based Automatic Style Matching Text-to-Speech Synthesis","date":"2025-04-14","arxiv_id":"2504.10309","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-autoregressive-neural-codec-language","title":"Pseudo-Autoregressive Neural Codec Language Models for Efficient Zero-Shot Text-to-Speech Synthesis","date":"2025-04-14","arxiv_id":"2504.10352","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-multilingual-text-to-speech","title":"Generalized Multilingual Text-to-Speech Generation with Language-Aware Style Adaptation","date":"2025-04-11","arxiv_id":"2504.08274","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-global-voices-a-data-efficient","title":"Empowering Global Voices: A Data-Efficient, Phoneme-Tone Adaptive Approach to High-Fidelity Speech Synthesis","date":"2025-04-10","arxiv_id":"2504.07858","repositories_listed":0,"syntology":null},{"url":null,"slug":"slimspeech-lightweight-and-efficient-text-to","title":"SlimSpeech: Lightweight and Efficient Text-to-Speech with Slim Rectified Flow","date":"2025-04-10","arxiv_id":"2504.07776","repositories_listed":0,"syntology":null},{"url":null,"slug":"speakeasy-enhancing-text-to-speech","title":"SpeakEasy: Enhancing Text-to-Speech Interactions for Expressive Content Creation","date":"2025-04-07","arxiv_id":"2504.05106","repositories_listed":0,"syntology":null},{"url":null,"slug":"speculative-end-turn-detector-for-efficient","title":"Speculative End-Turn Detector for Efficient Speech Chatbot Assistant","date":"2025-03-30","arxiv_id":"2503.23439","repositories_listed":0,"syntology":null},{"url":null,"slug":"supertonictts-towards-highly-scalable-and","title":"SupertonicTTS: Towards Highly Scalable and Efficient Text-to-Speech System","date":"2025-03-29","arxiv_id":"2503.23108","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepaudio-v1-towards-multi-modal-multi-stage","title":"DeepAudio-V1:Towards Multi-Modal Multi-Stage End-to-End Video to Speech and Audio Generation","date":"2025-03-28","arxiv_id":"2503.22265","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-audio-centric-modality-coupling-for","title":"Dual Audio-Centric Modality Coupling for Talking Head Generation","date":"2025-03-26","arxiv_id":"2503.22728","repositories_listed":0,"syntology":null},{"url":null,"slug":"your-voice-is-your-voice-supporting-self","title":"Your voice is your voice: Supporting Self-expression through Speech Generation and LLMs in Augmented and Alternative Communication","date":"2025-03-21","arxiv_id":"2503.17479","repositories_listed":0,"syntology":null},{"url":null,"slug":"mavflow-preserving-paralinguistic-elements","title":"MAVFlow: Preserving Paralinguistic Elements with Conditional Flow Matching for Zero-Shot AV2AV Multilingual Translation","date":"2025-03-14","arxiv_id":"2503.11026","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exhaustive-evaluation-of-tts-and-vc-based","title":"An Exhaustive Evaluation of TTS- and VC-based Data Augmentation for ASR","date":"2025-03-11","arxiv_id":"2503.08954","repositories_listed":0,"syntology":null},{"url":null,"slug":"vocaleyes-enhancing-environmental-perception","title":"VocalEyes: Enhancing Environmental Perception for the Visually Impaired through Vision-Language Models and Distance-Aware Object Detection","date":"2025-03-10","arxiv_id":"2503.16488","repositories_listed":0,"syntology":null},{"url":null,"slug":"inserter-speech-instruction-following-with","title":"InSerter: Speech Instruction Following with Unsupervised Interleaved Pre-training","date":"2025-03-04","arxiv_id":"2503.02769","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-speech-to-speech-translation-a-review","title":"Direct Speech to Speech Translation: A Review","date":"2025-03-03","arxiv_id":"2503.04799","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniwav-towards-unified-pre-training-for","title":"UniWav: Towards Unified Pre-training for Speech Representation Learning and Generation","date":"2025-03-02","arxiv_id":"2503.00733","repositories_listed":0,"syntology":null},{"url":null,"slug":"telephone-surveys-meet-conversational-ai","title":"Telephone Surveys Meet Conversational AI: Evaluating a LLM-Based Telephone Survey System at Scale","date":"2025-02-27","arxiv_id":"2502.20140","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-tts-contrastive-text-content-and-mel","title":"Clip-TTS: Contrastive Text-content and Mel-spectrogram, A High-Quality Text-to-Speech Method based on Contextual Semantic Understanding","date":"2025-02-26","arxiv_id":"2502.18889","repositories_listed":0,"syntology":null},{"url":null,"slug":"nexus-o-an-omni-perceptive-and-interactive","title":"Nexus: An Omni-Perceptive And -Interactive Model for Language, Audio, And Vision","date":"2025-02-26","arxiv_id":"2503.01879","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-alignment-enhanced-latent-diffusion","title":"MegaTTS 3: Sparse Alignment Enhanced Latent Diffusion Transformer for Zero-Shot Speech Synthesis","date":"2025-02-26","arxiv_id":"2502.18924","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-speech-understanding-and-generation","title":"Balancing Speech Understanding and Generation Using Continual Pre-training for Codec-based Speech LLM","date":"2025-02-24","arxiv_id":"2502.16897","repositories_listed":0,"syntology":null},{"url":null,"slug":"naturall2s-end-to-end-high-quality","title":"NaturalL2S: End-to-End High-quality Multispeaker Lip-to-Speech Synthesis with Differential Digital Signal Processing","date":"2025-02-17","arxiv_id":"2502.12002","repositories_listed":0,"syntology":null},{"url":null,"slug":"syncspeech-low-latency-and-efficient-dual","title":"SyncSpeech: Low-Latency and Efficient Dual-Stream Text-to-Speech based on Temporal Masked Transformer","date":"2025-02-16","arxiv_id":"2502.11094","repositories_listed":0,"syntology":null},{"url":null,"slug":"asvspoof-5-design-collection-and-validation","title":"ASVspoof 5: Design, Collection and Validation of Resources for Spoofing, Deepfake, and Adversarial Attack Detection Using Crowdsourced Speech","date":"2025-02-13","arxiv_id":"2502.08857","repositories_listed":0,"syntology":null}],"record_sha256":"df1d7fb62218ccf79e2ef902726f0f38f55d212376a6d15e844b1e933e222592","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}