{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/5","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":13,"rows_per_page":100,"rows":[401,500],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/4","next":"/task/speech-synthesis/papers/6","papers":[{"url":null,"slug":"aligndit-multimodal-aligned-diffusion","title":"AlignDiT: Multimodal Aligned Diffusion Transformer for Synchronized Speech Generation","date":"2025-04-29","arxiv_id":"2504.20629","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-flow-matching-based-tts-without","title":"Towards Flow-Matching-based TTS without Classifier-Free Guidance","date":"2025-04-29","arxiv_id":"2504.20334","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-network-based-voice","title":"Generative Adversarial Network based Voice Conversion: Techniques, Challenges, and Recent Advancements","date":"2025-04-27","arxiv_id":"2504.19197","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-framework-for-automated-1","title":"A Multi-Agent Framework for Automated Qinqiang Opera Script Generation Using Large Language Models","date":"2025-04-22","arxiv_id":"2504.15552","repositories_listed":0,"syntology":null},{"url":null,"slug":"fadel-uncertainty-aware-fake-audio-detection","title":"FADEL: Uncertainty-aware Fake Audio Detection with Evidential Deep Learning","date":"2025-04-22","arxiv_id":"2504.15663","repositories_listed":0,"syntology":null},{"url":null,"slug":"solido-a-robust-watermarking-method-for","title":"SOLIDO: A Robust Watermarking Method for Speech Synthesis via Low-Rank Adaptation","date":"2025-04-21","arxiv_id":"2504.15035","repositories_listed":0,"syntology":null},{"url":null,"slug":"collective-learning-mechanism-based-optimal","title":"Collective Learning Mechanism based Optimal Transport Generative Adversarial Network for Non-parallel Voice Conversion","date":"2025-04-18","arxiv_id":"2504.13791","repositories_listed":0,"syntology":null},{"url":null,"slug":"autostyle-tts-retrieval-augmented-generation","title":"AutoStyle-TTS: Retrieval-Augmented Generation based Automatic Style Matching Text-to-Speech Synthesis","date":"2025-04-14","arxiv_id":"2504.10309","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-autoregressive-neural-codec-language","title":"Pseudo-Autoregressive Neural Codec Language Models for Efficient Zero-Shot Text-to-Speech Synthesis","date":"2025-04-14","arxiv_id":"2504.10352","repositories_listed":0,"syntology":null},{"url":null,"slug":"amnet-an-acoustic-model-network-for-enhanced","title":"AMNet: An Acoustic Model Network for Enhanced Mandarin Speech Synthesis","date":"2025-04-12","arxiv_id":"2504.09225","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-global-voices-a-data-efficient","title":"Empowering Global Voices: A Data-Efficient, Phoneme-Tone Adaptive Approach to High-Fidelity Speech Synthesis","date":"2025-04-10","arxiv_id":"2504.07858","repositories_listed":0,"syntology":null},{"url":null,"slug":"slimspeech-lightweight-and-efficient-text-to","title":"SlimSpeech: Lightweight and Efficient Text-to-Speech with Slim Rectified Flow","date":"2025-04-10","arxiv_id":"2504.07776","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicecraft-dub-automated-video-dubbing-with","title":"VoiceCraft-Dub: Automated Video Dubbing with Neural Codec Language Models","date":"2025-04-03","arxiv_id":"2504.02386","repositories_listed":0,"syntology":null},{"url":null,"slug":"supertonictts-towards-highly-scalable-and","title":"SupertonicTTS: Towards Highly Scalable and Efficient Text-to-Speech System","date":"2025-03-29","arxiv_id":"2503.23108","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-faces-to-voices-learning-hierarchical","title":"From Faces to Voices: Learning Hierarchical Representations for High-quality Video-to-Speech","date":"2025-03-21","arxiv_id":"2503.16956","repositories_listed":0,"syntology":null},{"url":null,"slug":"good-practices-for-evaluation-of-synthesized","title":"Good practices for evaluation of synthesized speech","date":"2025-03-05","arxiv_id":"2503.03250","repositories_listed":0,"syntology":null},{"url":null,"slug":"2503-01266","title":"Voice Cloning for Dysarthric Speech Synthesis: Addressing Data Scarcity in Speech-Language Pathology","date":"2025-03-03","arxiv_id":"2503.01266","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffcss-diverse-and-expressive-conversational","title":"DiffCSS: Diverse and Expressive Conversational Speech Synthesis with Diffusion Models","date":"2025-02-27","arxiv_id":"2502.19924","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-alignment-enhanced-latent-diffusion","title":"MegaTTS 3: Sparse Alignment Enhanced Latent Diffusion Transformer for Zero-Shot Speech Synthesis","date":"2025-02-26","arxiv_id":"2502.18924","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-speech-understanding-and-generation","title":"Balancing Speech Understanding and Generation Using Continual Pre-training for Codec-based Speech LLM","date":"2025-02-24","arxiv_id":"2502.16897","repositories_listed":0,"syntology":null},{"url":null,"slug":"av-flow-transforming-text-to-audio-visual","title":"AV-Flow: Transforming Text to Audio-Visual Human-like Interactions","date":"2025-02-18","arxiv_id":"2502.13133","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-fidelity-music-vocoder-using-neural","title":"High-Fidelity Music Vocoder using Neural Audio Codecs","date":"2025-02-18","arxiv_id":"2502.12759","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-bridging-eeg-signals-and","title":"A Survey on Bridging EEG Signals and Generative AI: From Image and Text to Beyond","date":"2025-02-17","arxiv_id":"2502.12048","repositories_listed":0,"syntology":null},{"url":null,"slug":"naturall2s-end-to-end-high-quality","title":"NaturalL2S: End-to-End High-quality Multispeaker Lip-to-Speech Synthesis with Differential Digital Signal Processing","date":"2025-02-17","arxiv_id":"2502.12002","repositories_listed":0,"syntology":null},{"url":null,"slug":"felle-autoregressive-speech-synthesis-with","title":"FELLE: Autoregressive Speech Synthesis with Token-Wise Coarse-to-Fine Flow Matching","date":"2025-02-16","arxiv_id":"2502.11128","repositories_listed":0,"syntology":null},{"url":null,"slug":"asvspoof-5-design-collection-and-validation","title":"ASVspoof 5: Design, Collection and Validation of Resources for Spoofing, Deepfake, and Adversarial Attack Detection Using Crowdsourced Speech","date":"2025-02-13","arxiv_id":"2502.08857","repositories_listed":0,"syntology":null},{"url":null,"slug":"lorp-tts-low-rank-personalized-text-to-speech","title":"LoRP-TTS: Low-Rank Personalized Text-To-Speech","date":"2025-02-11","arxiv_id":"2502.07562","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-invasive-electromyographic-speech","title":"Non-invasive electromyographic speech neuroprosthesis: a geometric perspective","date":"2025-02-09","arxiv_id":"2502.05762","repositories_listed":0,"syntology":null},{"url":null,"slug":"gender-bias-in-instruction-guided-speech","title":"Gender Bias in Instruction-Guided Speech Synthesis Models","date":"2025-02-08","arxiv_id":"2502.05649","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-autoregressive-modeling-with","title":"Continuous Autoregressive Modeling with Stochastic Monotonic Alignment for Speech Synthesis","date":"2025-02-03","arxiv_id":"2502.01084","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-neural-tts-voices-for-accessibility","title":"Compact Neural TTS Voices for Accessibility","date":"2025-01-28","arxiv_id":"2501.17332","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-audio-deepfake-detection-via","title":"Generalizable Audio Deepfake Detection via Latent Space Refinement and Augmentation","date":"2025-01-24","arxiv_id":"2501.14240","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-data-augmentation-challenge-zero","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","date":"2025-01-23","arxiv_id":"2501.13372","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-non-autoregressive-model-for-joint-stt-and","title":"A Non-autoregressive Model for Joint STT and TTS","date":"2025-01-15","arxiv_id":"2501.09104","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-along-perceptual-voice","title":"Speech Synthesis along Perceptual Voice Quality Dimensions","date":"2025-01-15","arxiv_id":"2501.08791","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-encoding-of-linguistic","title":"Exploring the encoding of linguistic representations in the Fully-Connected Layer of generative CNNs for Speech","date":"2025-01-13","arxiv_id":"2501.07726","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-text-to-speech-synthesis-using","title":"Low-Resource Text-to-Speech Synthesis Using Noise-Augmented Training of ForwardTacotron","date":"2025-01-10","arxiv_id":"2501.05976","repositories_listed":0,"syntology":null},{"url":null,"slug":"proemo-prompt-driven-text-to-speech-synthesis","title":"PROEMO: Prompt-Driven Text-to-Speech Synthesis Based on Emotion and Intensity Control","date":"2025-01-10","arxiv_id":"2501.06276","repositories_listed":0,"syntology":null},{"url":null,"slug":"tts-transducer-end-to-end-speech-synthesis","title":"TTS-Transducer: End-to-End Speech Synthesis with Neural Transducer","date":"2025-01-10","arxiv_id":"2501.06320","repositories_listed":0,"syntology":null},{"url":null,"slug":"jelly-joint-emotion-recognition-and-context","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","date":"2025-01-09","arxiv_id":"2501.04904","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-speaker-specific-features-in-speaker","title":"Probing Speaker-specific Features in Speaker Representations","date":"2025-01-09","arxiv_id":"2501.05310","repositories_listed":0,"syntology":null},{"url":null,"slug":"flespeech-flexibly-controllable-speech","title":"FleSpeech: Flexibly Controllable Speech Generation with Various Prompts","date":"2025-01-08","arxiv_id":"2501.04644","repositories_listed":0,"syntology":null},{"url":"/paper/facespeak-expressive-and-high-quality-speech","slug":"facespeak-expressive-and-high-quality-speech","title":"FaceSpeak: Expressive and High-Quality Speech Synthesis from Human Portraits of Different Styles","date":"2025-01-02","arxiv_id":"2501.03181","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/facespeak-expressive-and-high-quality-speech#ran","syntology_url":"https://syntology.ai/paper/2501.03181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03181"}},"official":null}},{"url":null,"slug":"crossspeech-cross-lingual-speech-synthesis","title":"CrossSpeech++: Cross-lingual Speech Synthesis with Decoupled Language and Speaker Generation","date":"2024-12-28","arxiv_id":"2412.20048","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-tts-stable-speaker-adaptive-text-to","title":"Stable-TTS: Stable Speaker-Adaptive Text-to-Speech Synthesis via Prosody Prompting","date":"2024-12-28","arxiv_id":"2412.20155","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicedit-dual-condition-diffusion-transformer","title":"VoiceDiT: Dual-Condition Diffusion Transformer for Environment-Aware Speech Synthesis","date":"2024-12-26","arxiv_id":"2412.19259","repositories_listed":0,"syntology":null},{"url":null,"slug":"intra-and-inter-modal-context-interaction","title":"Intra- and Inter-modal Context Interaction Modeling for Conversational Speech Synthesis","date":"2024-12-25","arxiv_id":"2412.18733","repositories_listed":0,"syntology":null},{"url":null,"slug":"mri2speech-speech-synthesis-from-articulatory","title":"MRI2Speech: Speech Synthesis from Articulatory Movements Recorded by Real-time MRI","date":"2024-12-25","arxiv_id":"2412.18836","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-speech-synthesis-with-next","title":"Autoregressive Speech Synthesis with Next-Distribution Prediction","date":"2024-12-22","arxiv_id":"2412.16846","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-disentanglement-for-environment","title":"Incremental Disentanglement for Environment-Aware Zero-Shot Text-to-Speech Synthesis","date":"2024-12-22","arxiv_id":"2412.16977","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-speech-synthesis-from-multimodal","title":"Deep Speech Synthesis from Multimodal Articulatory Representations","date":"2024-12-17","arxiv_id":"2412.13387","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodyfm-unsupervised-phrasing-and","title":"ProsodyFM: Unsupervised Phrasing and Intonation Control for Intelligible Speech Synthesis","date":"2024-12-16","arxiv_id":"2412.11795","repositories_listed":0,"syntology":null},{"url":null,"slug":"amused-an-attentive-deep-neural-network-for","title":"AMuSeD: An Attentive Deep Neural Network for Multimodal Sarcasm Detection Incorporating Bi-modal Data Augmentation","date":"2024-12-13","arxiv_id":"2412.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-generative-modeling-with-residual","title":"Efficient Generative Modeling with Residual Vector Quantization-Based Tokens","date":"2024-12-13","arxiv_id":"2412.10208","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-mono-to-binaural-speech-synthesis","title":"Zero-Shot Mono-to-Binaural Speech Synthesis","date":"2024-12-11","arxiv_id":"2412.08356","repositories_listed":0,"syntology":null},{"url":null,"slug":"analytic-study-of-text-free-speech-synthesis","title":"Analytic Study of Text-Free Speech Synthesis for Raw Audio using a Self-Supervised Learning Model","date":"2024-12-04","arxiv_id":"2412.03074","repositories_listed":0,"syntology":null},{"url":null,"slug":"visatronic-a-multimodal-decoder-only-model","title":"Visatronic: A Multimodal Decoder-Only Model for Speech Synthesis","date":"2024-11-26","arxiv_id":"2411.17690","repositories_listed":0,"syntology":null},{"url":null,"slug":"vqalattent-a-transparent-speech-generation","title":"VQalAttent: a Transparent Speech Generation Pipeline based on Transformer-learned VQ-VAE Latent Space","date":"2024-11-22","arxiv_id":"2411.14642","repositories_listed":0,"syntology":null},{"url":null,"slug":"debatts-zero-shot-debating-text-to-speech","title":"Debatts: Zero-Shot Debating Text-to-Speech Synthesis","date":"2024-11-10","arxiv_id":"2411.06540","repositories_listed":0,"syntology":null},{"url":null,"slug":"complete-reconstruction-of-the-tongue-contour","title":"Complete reconstruction of the tongue contour through acoustic to articulatory inversion using real-time MRI data","date":"2024-11-04","arxiv_id":"2411.02037","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-polish-automatic-speech","title":"Augmenting Polish Automatic Speech Recognition System With Synthetic Data","date":"2024-10-30","arxiv_id":"2410.22903","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-high-quality-auto-regressive-speech","title":"Fast and High-Quality Auto-Regressive Speech Synthesis via Speculative Decoding","date":"2024-10-29","arxiv_id":"2410.21951","repositories_listed":0,"syntology":null},{"url":null,"slug":"get-large-language-models-ready-to-speak-a","title":"Get Large Language Models Ready to Speak: A Late-fusion Approach for Speech Generation","date":"2024-10-27","arxiv_id":"2410.20336","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-social-platforms-accessible-emotion","title":"Making Social Platforms Accessible: Emotion-Aware Speech Generation with Integrated Text Analysis","date":"2024-10-24","arxiv_id":"2410.19199","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-speech-synthesis-using-per-token","title":"Continuous Speech Synthesis using per-token Latent Diffusion","date":"2024-10-21","arxiv_id":"2410.16048","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-collecting-text-to","title":"A Unified Framework for Collecting Text-to-Speech Synthesis Datasets for 22 Indian Languages","date":"2024-10-18","arxiv_id":"2410.14197","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-codec-based-speech-synthesis","title":"Accelerating Codec-based Speech Synthesis with Multi-Token Prediction and Speculative Decoding","date":"2024-10-17","arxiv_id":"2410.13839","repositories_listed":0,"syntology":null},{"url":null,"slug":"dart-disentanglement-of-accent-and-speaker","title":"DART: Disentanglement of Accent and Speaker Representation in Multispeaker Text-to-Speech","date":"2024-10-17","arxiv_id":"2410.13342","repositories_listed":0,"syntology":null},{"url":null,"slug":"durian-e-2-duration-informed-attention","title":"DurIAN-E 2: Duration Informed Attention Network with Adaptive Variational Autoencoder and Adversarial Learning for Expressive Text-to-Speech Synthesis","date":"2024-10-17","arxiv_id":"2410.13288","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-oversmoothing-evaluating-ddpm-and-mse","title":"Beyond Oversmoothing: Evaluating DDPM and MSE for Scalable Speech Synthesis in ASR","date":"2024-10-16","arxiv_id":"2410.12279","repositories_listed":0,"syntology":null},{"url":null,"slug":"dmdspeech-distilled-diffusion-model","title":"DMOSpeech: Direct Metric Optimization via Distilled Diffusion Model in Zero-Shot Speech Synthesis","date":"2024-10-14","arxiv_id":"2410.11097","repositories_listed":0,"syntology":null},{"url":null,"slug":"everyday-speech-in-the-indian-subcontinent","title":"Everyday Speech in the Indian Subcontinent","date":"2024-10-14","arxiv_id":"2410.10508","repositories_listed":0,"syntology":null},{"url":null,"slug":"bahasa-harmony-a-comprehensive-dataset-for","title":"Bahasa Harmony: A Comprehensive Dataset for Bahasa Text-to-Speech Synthesis with Discrete Codec Modeling of EnGen-TTS","date":"2024-10-09","arxiv_id":"2410.06608","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-training-strategies-for-natural","title":"Efficient training strategies for natural sounding speech synthesis and speaker adaptation based on FastPitch","date":"2024-10-09","arxiv_id":"2410.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"hall-e-hierarchical-neural-codec-language","title":"HALL-E: Hierarchical Neural Codec Language Model for Minute-Long Zero-Shot Text-to-Speech Synthesis","date":"2024-10-06","arxiv_id":"2410.04380","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-and-robust-defenses-in","title":"Adversarial Attacks and Robust Defenses in Speaker Embedding based Zero-Shot Text-to-Speech System","date":"2024-10-05","arxiv_id":"2410.04017","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-semantic-communication-for-text-to","title":"Generative Semantic Communication for Text-to-Speech Synthesis","date":"2024-10-04","arxiv_id":"2410.03459","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiverse-efficient-and-expressive-zero-shot","title":"MultiVerse: Efficient and Expressive Zero-Shot Multi-Task Text-to-Speech","date":"2024-10-04","arxiv_id":"2410.03192","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-conversion-using-discrete-units-with","title":"Accent conversion using discrete units with parallel data synthesized from controllable accented TTS","date":"2024-09-30","arxiv_id":"2410.03734","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-analysis-of-audio-visual-tasks","title":"Quantitative Analysis of Audio-Visual Tasks: An Information-Theoretic Perspective","date":"2024-09-29","arxiv_id":"2409.19575","repositories_listed":0,"syntology":null},{"url":null,"slug":"emopro-a-prompt-selection-strategy-for","title":"EmoPro: A Prompt Selection Strategy for Emotional Expression in LM-based Speech Synthesis","date":"2024-09-27","arxiv_id":"2409.18512","repositories_listed":0,"syntology":null},{"url":null,"slug":"facial-expression-enhanced-tts-combining-face","title":"Facial Expression-Enhanced TTS: Combining Face Representation and Emotion Intensity for Adaptive Speech","date":"2024-09-24","arxiv_id":"2409.16203","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylefusion-tts-multimodal-style-control-and","title":"StyleFusion TTS: Multimodal Style-control and Enhanced Feature Fusion for Zero-shot Text-to-speech Synthesis","date":"2024-09-24","arxiv_id":"2409.15741","repositories_listed":0,"syntology":null},{"url":null,"slug":"ndvq-robust-neural-audio-codec-with-normal","title":"NDVQ: Robust Neural Audio Codec with Normal Distribution-Based Vector Quantization","date":"2024-09-19","arxiv_id":"2409.12717","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multilingual-speech-generation-and","title":"Enhancing Multilingual Speech Generation and Recognition Abilities in LLMs with Constructed Code-switched Data","date":"2024-09-17","arxiv_id":"2409.10969","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-stage-tts-with-masked-audio-token","title":"Single-stage TTS with Masked Audio Token Modeling and Semantic Knowledge Distillation","date":"2024-09-17","arxiv_id":"2409.11003","repositories_listed":0,"syntology":null},{"url":null,"slug":"emo-dpo-controllable-emotional-speech","title":"Emo-DPO: Controllable Emotional Speech Synthesis through Direct Preference Optimization","date":"2024-09-16","arxiv_id":"2409.10157","repositories_listed":0,"syntology":null},{"url":null,"slug":"styletts-zs-efficient-high-quality-zero-shot","title":"StyleTTS-ZS: Efficient High-Quality Zero-Shot Text-to-Speech Synthesis with Distilled Time-Varying Style Diffusion","date":"2024-09-16","arxiv_id":"2409.10058","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-robustness-of-diffusion-based-zero","title":"Improving Robustness of Diffusion-Based Zero-Shot Speech Synthesis via Stable Formant Generation","date":"2024-09-14","arxiv_id":"2409.09311","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-powered-grapheme-to-phoneme-conversion","title":"LLM-Powered Grapheme-to-Phoneme Conversion: Benchmark and Case Study","date":"2024-09-13","arxiv_id":"2409.08554","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-in-the-wild","title":"Text-To-Speech Synthesis In The Wild","date":"2024-09-13","arxiv_id":"2409.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-text-error-correction-for-chinese-speech","title":"Full-text Error Correction for Chinese Speech Recognition with Large Language Model","date":"2024-09-12","arxiv_id":"2409.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"2409-13734","title":"Enhancing Kurdish Text-to-Speech with Native Corpus Training: A High-Quality WaveGlow Vocoder Approach","date":"2024-09-10","arxiv_id":"2409.13734","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-happens-to-diffusion-model-likelihood","title":"What happens to diffusion model likelihood when your model is conditional?","date":"2024-09-10","arxiv_id":"2409.06364","repositories_listed":0,"syntology":null},{"url":null,"slug":"as-speech-adaptive-style-for-speech-synthesis","title":"AS-Speech: Adaptive Style For Speech Synthesis","date":"2024-09-09","arxiv_id":"2409.05730","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-high-quality-and-parameter-efficient","title":"Fast, High-Quality and Parameter-Efficient Articulatory Synthesis using Differentiable DSP","date":"2024-09-04","arxiv_id":"2409.02451","repositories_listed":0,"syntology":null},{"url":null,"slug":"vec2wav-2-0-advancing-voice-conversion-via","title":"vec2wav 2.0: Advancing Voice Conversion via Discrete Token Vocoders","date":"2024-09-03","arxiv_id":"2409.01995","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxhakka-a-dialectally-diverse-multi-speaker","title":"VoxHakka: A Dialectally Diverse Multi-speaker Text-to-Speech System for Taiwanese Hakka","date":"2024-09-03","arxiv_id":"2409.01548","repositories_listed":0,"syntology":null},{"url":null,"slug":"selecttts-synthesizing-anyone-s-voice-via","title":"SelectTTS: Synthesizing Anyone's Voice via Discrete Unit-Based Frame Selection","date":"2024-08-30","arxiv_id":"2408.17432","repositories_listed":0,"syntology":null},{"url":null,"slug":"literary-and-colloquial-dialect","title":"Literary and Colloquial Dialect Identification for Tamil using Acoustic Features","date":"2024-08-27","arxiv_id":"2408.14887","repositories_listed":0,"syntology":null}],"record_sha256":"4bed8d8fe322c7f53f656f2d644b43ea952b6d2bc1d648fa95f4752d9dd5560a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}