{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-1/papers/8","list_of":"/task/text-to-speech-1","task":"text-to-speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":15,"rows_per_page":100,"rows":[701,800],"of":1413,"counts":{"archive_papers_tagged":1413,"with_a_code_link":395,"where_syntology_ran_a_sample":106,"not_listed_spam_title":0,"listed":1413,"listed_where_code_ran":106,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-1","prev":"/task/text-to-speech-1/papers/7","next":"/task/text-to-speech-1/papers/9","papers":[{"url":null,"slug":"multi-speaker-text-to-speech-training-with","title":"Multi-speaker Text-to-speech Training with Speaker Anonymized Data","date":"2024-05-20","arxiv_id":"2405.11767","repositories_listed":0,"syntology":null},{"url":null,"slug":"vr-gpt-visual-language-model-for-intelligent","title":"VR-GPT: Visual Language Model for Intelligent Virtual Reality Applications","date":"2024-05-19","arxiv_id":"2405.11537","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speech-style-spaces-with-language","title":"Exploring speech style spaces with language models: Emotional TTS without emotion labels","date":"2024-05-18","arxiv_id":"2405.11413","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-luganda-text-to-speech-model-from","title":"Building a Luganda Text-to-Speech Model From Crowdsourced Data","date":"2024-05-16","arxiv_id":"2405.10211","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-text-to-speech-synthesis-from-a","title":"Evaluating Text-to-Speech Synthesis from a Large Discrete Token-based Speech Language Model","date":"2024-05-16","arxiv_id":"2405.09768","repositories_listed":0,"syntology":null},{"url":null,"slug":"faces-that-speak-jointly-synthesising-talking","title":"Faces that Speak: Jointly Synthesising Talking Face and Speech from Text","date":"2024-05-16","arxiv_id":"2405.10272","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-evaluating-the-robustness-of","title":"Towards Evaluating the Robustness of Automatic Speech Recognition Systems via Audio Style Transfer","date":"2024-05-15","arxiv_id":"2405.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-pill-identification-for-the","title":"Real-Time Pill Identification for the Visually Impaired Using Deep Learning","date":"2024-05-08","arxiv_id":"2405.05983","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-constrained-inference-for-robust","title":"Attention-Constrained Inference for Robust Decoder-Only Text-to-Speech","date":"2024-04-30","arxiv_id":"2404.19723","repositories_listed":0,"syntology":null},{"url":null,"slug":"ti-asu-toward-robust-automatic-speech","title":"TI-ASU: Toward Robust Automatic Speech Understanding through Text-to-speech Imputation Against Missing Speech Modality","date":"2024-04-27","arxiv_id":"2404.17983","repositories_listed":0,"syntology":null},{"url":null,"slug":"storytts-a-highly-expressive-text-to-speech","title":"StoryTTS: A Highly Expressive Text-to-Speech Dataset with Rich Textual Expressiveness Annotations","date":"2024-04-23","arxiv_id":"2404.14946","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-audio-deepfake-detection","title":"Retrieval-Augmented Audio Deepfake Detection","date":"2024-04-22","arxiv_id":"2404.13892","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-agnostic-multi-scale-contrastive-text","title":"Prior-agnostic Multi-scale Contrastive Text-Audio Pre-training for Parallelized TTS Frontend Modeling","date":"2024-04-14","arxiv_id":"2404.09192","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-assisted-real-time-traffic-sign","title":"Voice-Assisted Real-Time Traffic Sign Recognition System Using Convolutional Neural Network","date":"2024-04-11","arxiv_id":"2404.07807","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-x-lance-technical-report-for-interspeech","title":"The X-LANCE Technical Report for Interspeech 2024 Speech Processing Using Discrete Speech Unit Challenge","date":"2024-04-09","arxiv_id":"2404.06079","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-domain-audio-deepfake-detection-dataset","title":"Cross-Domain Audio Deepfake Detection: Dataset and Analysis","date":"2024-04-07","arxiv_id":"2404.04904","repositories_listed":0,"syntology":null},{"url":null,"slug":"rall-e-robust-codec-language-modeling-with","title":"RALL-E: Robust Codec Language Modeling with Chain-of-Thought Prompting for Text-to-Speech Synthesis","date":"2024-04-04","arxiv_id":"2404.03204","repositories_listed":0,"syntology":null},{"url":null,"slug":"clam-tts-improving-neural-codec-language","title":"CLaM-TTS: Improving Neural Codec Language Model for Zero-Shot Text-to-Speech","date":"2024-04-03","arxiv_id":"2404.02781","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-multi-modal-large-language-and","title":"A Review of Multi-Modal Large Language and Vision Models","date":"2024-03-28","arxiv_id":"2404.01322","repositories_listed":0,"syntology":null},{"url":null,"slug":"isometric-neural-machine-translation-using","title":"Isometric Neural Machine Translation using Phoneme Count Ratio Reward-based Reinforcement Learning","date":"2024-03-20","arxiv_id":"2403.15469","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-an-african-american-sounding-tts","title":"Creating an African American-Sounding TTS: Guidelines, Technical Challenges,and Surprising Evaluations","date":"2024-03-17","arxiv_id":"2403.11209","repositories_listed":0,"syntology":null},{"url":null,"slug":"em-tts-efficiently-trained-low-resource","title":"EM-TTS: Efficiently Trained Low-Resource Mongolian Lightweight Text-to-Speech","date":"2024-03-13","arxiv_id":"2403.08164","repositories_listed":0,"syntology":null},{"url":null,"slug":"attempt-towards-stress-transfer-in-speech-to","title":"Attempt Towards Stress Transfer in Speech-to-Speech Machine Translation","date":"2024-03-07","arxiv_id":"2403.04178","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentionstitch-how-attention-solves-the","title":"AttentionStitch: How Attention Solves the Speech Editing Problem","date":"2024-03-05","arxiv_id":"2403.04804","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-accurate-lip-to-speech-synthesis-in","title":"Towards Accurate Lip-to-Speech Synthesis in-the-Wild","date":"2024-03-02","arxiv_id":"2403.01087","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-multilingual-speech-synthesis-to","title":"Extending Multilingual Speech Synthesis to 100+ Languages without Transcribed Data","date":"2024-02-29","arxiv_id":"2402.18932","repositories_listed":0,"syntology":null},{"url":null,"slug":"daisy-tts-simulating-wider-spectrum-of","title":"Daisy-TTS: Simulating Wider Spectrum of Emotions via Prosody Embedding Decomposition","date":"2024-02-22","arxiv_id":"2402.14523","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-data-selection-employing-semantic","title":"Efficient data selection employing Semantic Similarity-based Graph Structures for model training","date":"2024-02-22","arxiv_id":"2402.14888","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-semantic-latent-space-of-diffusion","title":"On the Semantic Latent Space of Diffusion-Based Text-to-Speech Models","date":"2024-02-19","arxiv_id":"2402.12423","repositories_listed":0,"syntology":null},{"url":null,"slug":"ain-t-misbehavin-using-llms-to-generate","title":"Ain't Misbehavin' -- Using LLMs to Generate Expressive Robot Behavior in Conversations with the Tabletop Robot Haru","date":"2024-02-18","arxiv_id":"2402.11571","repositories_listed":0,"syntology":null},{"url":"/paper/mobilespeech-a-fast-and-high-fidelity","slug":"mobilespeech-a-fast-and-high-fidelity","title":"MobileSpeech: A Fast and High-Fidelity Framework for Mobile Zero-Shot Text-to-Speech","date":"2024-02-14","arxiv_id":"2402.09378","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":6,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mobilespeech-a-fast-and-high-fidelity#ran","syntology_url":"https://syntology.ai/paper/2402.09378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09378"}},"official":null}},{"url":null,"slug":"base-tts-lessons-from-building-a-billion","title":"BASE TTS: Lessons from building a billion-parameter Text-to-Speech model on 100K hours of data","date":"2024-02-12","arxiv_id":"2402.08093","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-flow-matching-based-zero-shot-text-to","title":"Making Flow-Matching-Based Zero-Shot Text-to-Speech Laugh as You Like","date":"2024-02-12","arxiv_id":"2402.07383","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-approach-to-voice-authenticity","title":"A New Approach to Voice Authenticity","date":"2024-02-09","arxiv_id":"2402.06304","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-the-stability-of-llm-based-speech","title":"Enhancing the Stability of LLM-based Speech Generation Systems through Self-Supervised Representations","date":"2024-02-05","arxiv_id":"2402.03407","repositories_listed":0,"syntology":null},{"url":null,"slug":"frame-wise-breath-detection-with-self","title":"Frame-Wise Breath Detection with Self-Training: An Exploration of Enhancing Breath Naturalness in Text-to-Speech","date":"2024-02-01","arxiv_id":"2402.00288","repositories_listed":0,"syntology":null},{"url":null,"slug":"muntts-a-text-to-speech-system-for-mundari","title":"MunTTS: A Text-to-Speech System for Mundari","date":"2024-01-28","arxiv_id":"2401.15579","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-t-decoder-only-generative-transducer-for","title":"VALL-T: Decoder-Only Generative Transducer for Robust and Decoding-Controllable Text-to-Speech","date":"2024-01-25","arxiv_id":"2401.14321","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximizing-data-efficiency-for-cross-lingual","title":"Maximizing Data Efficiency for Cross-Lingual TTS Adaptation by Self-Supervised Representation Mixing and Embedding Initialization","date":"2024-01-23","arxiv_id":"2402.01692","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-accessibility-voice-cloning-and","title":"Empowering Communication: Speech Technology for Indian and Western Accents through AI-powered Speech Synthesis","date":"2024-01-22","arxiv_id":"2401.11771","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-speech-for-voice-privacy","title":"Adversarial speech for voice privacy protection from Personalized Speech generation","date":"2024-01-22","arxiv_id":"2401.11857","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-grapheme-to-phoneme","title":"Data-driven grapheme-to-phoneme representations for a lexicon-free text-to-speech","date":"2024-01-19","arxiv_id":"2401.10465","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcmchaos-improvising-rap-music-with-mcmc","title":"MCMChaos: Improvising Rap Music with MCMC Methods and Chaos Theory","date":"2024-01-15","arxiv_id":"2401.07967","repositories_listed":0,"syntology":null},{"url":null,"slug":"ella-v-stable-neural-codec-language-modeling","title":"ELLA-V: Stable Neural Codec Language Modeling with Alignment-guided Sequence Reordering","date":"2024-01-14","arxiv_id":"2401.07333","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-hindi-to-english-speech-conversion","title":"End to end Hindi to English speech conversion using Bark, mBART and a finetuned XLSR Wav2Vec2","date":"2024-01-11","arxiv_id":"2401.06183","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-zero-shot-text-to-speech","title":"Noise-robust zero-shot text-to-speech synthesis conditioned on self-supervised speech-representation model with adapters","date":"2024-01-10","arxiv_id":"2401.05111","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-personalizing-user-perceived","title":"Evaluating and Personalizing User-Perceived Quality of Text-to-Speech Voices for Delivering Mindfulness Meditation with Different Physical Embodiments","date":"2024-01-07","arxiv_id":"2401.03581","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-the-linguistic-representations-from","title":"Transfer the linguistic representations from TTS to accent conversion with non-parallel data","date":"2024-01-07","arxiv_id":"2401.03538","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-fastpitch-chunk-based-high","title":"Incremental FastPitch: Chunk-based High Quality Text to Speech","date":"2024-01-03","arxiv_id":"2401.01755","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-neural-transducers-for-two-stage","title":"Utilizing Neural Transducers for Two-Stage Text-to-Speech via Semantic Token Prediction","date":"2024-01-03","arxiv_id":"2401.01498","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-large-language-model-for-speech","title":"Boosting Large Language Model for Speech Synthesis: An Empirical Study","date":"2023-12-30","arxiv_id":"2401.00246","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalization-of-lithuanian-text-using","title":"Normalization of Lithuanian Text Using Regular Expressions","date":"2023-12-29","arxiv_id":"2312.17660","repositories_listed":0,"syntology":null},{"url":null,"slug":"ae-flow-autoencoder-normalizing-flow","title":"AE-Flow: AutoEncoder Normalizing Flow","date":"2023-12-27","arxiv_id":"2312.16552","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-new-voices-using-normalizing-flows","title":"Creating New Voices using Normalizing Flows","date":"2023-12-22","arxiv_id":"2312.14569","repositories_listed":0,"syntology":null},{"url":null,"slug":"external-knowledge-augmented-polyphone","title":"External Knowledge Augmented Polyphone Disambiguation Using Large Language Model","date":"2023-12-19","arxiv_id":"2312.11920","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-based-study-on-different-text-to","title":"A review-based study on different Text-to-Speech technologies","date":"2023-12-17","arxiv_id":"2312.11563","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-tts-multi-modal-prompt-based-style","title":"MM-TTS: Multi-modal Prompt based Style Transfer for Expressive Text-to-Speech Synthesis","date":"2023-12-17","arxiv_id":"2312.10687","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-study-assessing-the-combined","title":"An Experimental Study: Assessing the Combined Framework of WavLM and BEST-RQ for Text-to-Speech Synthesis","date":"2023-12-08","arxiv_id":"2312.05415","repositories_listed":0,"syntology":null},{"url":null,"slug":"schrodinger-bridges-beat-diffusion-models-on","title":"Schrodinger Bridges Beat Diffusion Models on Text-to-Speech Synthesis","date":"2023-12-06","arxiv_id":"2312.03491","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-mixed-text-to-speech-synthesis-under-low","title":"Code-Mixed Text to Speech Synthesis under Low-Resource Constraints","date":"2023-12-02","arxiv_id":"2312.01103","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapid-speaker-adaptation-in-low-resource-text","title":"Rapid Speaker Adaptation in Low Resource Text to Speech Systems using Synthetic Data and Transfer learning","date":"2023-12-02","arxiv_id":"2312.01107","repositories_listed":0,"syntology":null},{"url":null,"slug":"vulnerability-of-automatic-identity","title":"Vulnerability of Automatic Identity Recognition to Audio-Visual Deepfakes","date":"2023-11-29","arxiv_id":"2311.17655","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-flows-for-generative-modeling-and","title":"Guided Flows for Generative Modeling and Decision Making","date":"2023-11-22","arxiv_id":"2311.13443","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-center-audio-video-intelligence-on","title":"Data Center Audio/Video Intelligence on Device (DAVID) -- An Edge-AI Platform for Smart-Toys","date":"2023-11-18","arxiv_id":"2311.11030","repositories_listed":0,"syntology":null},{"url":null,"slug":"utilizing-speech-emotion-recognition-and","title":"Utilizing Speech Emotion Recognition and Recommender Systems for Negative Emotion Handling in Therapy Chatbots","date":"2023-11-18","arxiv_id":"2311.11116","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-altering-the-latent-space-of","title":"A Study on Altering the Latent Space of Pretrained Text to Speech Models for Improved Expressiveness","date":"2023-11-17","arxiv_id":"2311.10804","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatanything-facetime-chat-with-llm-enhanced","title":"ChatAnything: Facetime Chat with LLM-Enhanced Personas","date":"2023-11-12","arxiv_id":"2311.06772","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-speaking-children-why-we-need-them","title":"Synthetic Speaking Children -- Why We Need Them and How to Make Them","date":"2023-11-08","arxiv_id":"2311.06307","repositories_listed":0,"syntology":null},{"url":null,"slug":"character-level-bangla-text-to-ipa","title":"Character-Level Bangla Text-to-IPA Transcription Using Transformer Architecture with Sequence Alignment","date":"2023-11-07","arxiv_id":"2311.03792","repositories_listed":0,"syntology":null},{"url":null,"slug":"transduce-and-speak-neural-transducer-for","title":"Transduce and Speak: Neural Transducer for Text-to-Speech with Semantic Token Prediction","date":"2023-11-06","arxiv_id":"2311.02898","repositories_listed":0,"syntology":null},{"url":null,"slug":"e3-tts-easy-end-to-end-diffusion-based-text","title":"E3 TTS: Easy End-to-End Diffusion-based Text to Speech","date":"2023-11-02","arxiv_id":"2311.00945","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-tts-driven-by-natural-language","title":"Expressive TTS Driven by Natural Language Prompts Using Few Human Annotations","date":"2023-11-02","arxiv_id":"2311.01260","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-description-based-text-to-speech-with","title":"Style Description based Text-to-Speech with Conditional Prosodic Layer Normalization based Diffusion GAN","date":"2023-10-27","arxiv_id":"2310.18169","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-pre-training-for-speech-with-flow","title":"Generative Pre-training for Speech with Flow Matching","date":"2023-10-25","arxiv_id":"2310.16338","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpp-tts-diversifying-prosodic-features-of-1","title":"DPP-TTS: Diversifying prosodic features of speech via determinantal point processes","date":"2023-10-23","arxiv_id":"2310.14663","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-text-to-speech-systems-and","title":"An overview of text-to-speech systems and media applications","date":"2023-10-22","arxiv_id":"2310.14301","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-relevance-of-phoneme-duration","title":"On the Relevance of Phoneme Duration Variability of Synthesized Training Data for Automatic Speech Recognition","date":"2023-10-12","arxiv_id":"2310.08132","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-transfer-learning-in","title":"Comparative Analysis of Transfer Learning in Deep Learning Text-to-Speech Models on a Few-Shot, Low-Resource, Customized Dataset","date":"2023-10-08","arxiv_id":"2310.04982","repositories_listed":0,"syntology":null},{"url":"/paper/neutral-tts-female-voice-corpus-in-brazilian","slug":"neutral-tts-female-voice-corpus-in-brazilian","title":"Neutral TTS Female Voice Corpus in Brazilian Portuguese","date":"2023-10-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/unified-speech-and-gesture-synthesis-using","slug":"unified-speech-and-gesture-synthesis-using","title":"Unified speech and gesture synthesis using flow matching","date":"2023-10-08","arxiv_id":"2310.05181","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-filling-latent-space-data-augmentation","title":"Latent Filling: Latent Space Data Augmentation for Zero-shot Speech Synthesis","date":"2023-10-05","arxiv_id":"2310.03538","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-voicemos-challenge-2023-zero-shot","title":"The VoiceMOS Challenge 2023: Zero-shot Subjective Speech Quality Prediction for Multiple Domains","date":"2023-10-04","arxiv_id":"2310.02640","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-human-like-spoken-dialogue-generation","title":"Towards human-like spoken dialogue generation between AI agents from written dialogue","date":"2023-10-02","arxiv_id":"2310.01088","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-self-supervised-learning-with","title":"Low-Resource Self-Supervised Learning with SSL-Enhanced TTS","date":"2023-09-29","arxiv_id":"2309.17020","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-speech-detection-based-on-temporal","title":"Synthetic Speech Detection Based on Temporal Consistency and Distribution of Speaker Features","date":"2023-09-29","arxiv_id":"2309.16954","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-fidelity-speech-synthesis-with-minimal","title":"High-Fidelity Speech Synthesis with Minimal Supervision: All Using Diffusion Models","date":"2023-09-27","arxiv_id":"2309.15512","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-stylespeech-improved-face-to-voice","title":"Face-StyleSpeech: Enhancing Zero-shot Speech Synthesis from Face Images with Improved Face-to-Speech Mapping","date":"2023-09-25","arxiv_id":"2311.05844","repositories_listed":0,"syntology":null},{"url":null,"slug":"voiceldm-text-to-speech-with-environmental","title":"VoiceLDM: Text-to-Speech with Environmental Context","date":"2023-09-24","arxiv_id":"2309.13664","repositories_listed":0,"syntology":null},{"url":null,"slug":"durian-e-duration-informed-attention-network","title":"DurIAN-E: Duration Informed Attention Network For Expressive Text-to-Speech Synthesis","date":"2023-09-22","arxiv_id":"2309.12792","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-silence-on-speech-anti-spoofing","title":"The Impact of Silence on Speech Anti-Spoofing","date":"2023-09-21","arxiv_id":"2309.11827","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-while-you-think-streaming-speech","title":"Speak While You Think: Streaming Speech Synthesis During Text Generation","date":"2023-09-20","arxiv_id":"2309.11210","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speech-enhancement-for-low-resource","title":"Exploring Speech Enhancement for Low-resource Speech Synthesis","date":"2023-09-19","arxiv_id":"2309.10795","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-speech-ptm-text-llm-and-emotional","title":"Leveraging Speech PTM, Text LLM, and Emotional TTS for Speech Emotion Recognition","date":"2023-09-19","arxiv_id":"2309.10294","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-text-for-spoken-language","title":"Augmenting text for spoken language understanding with Large Language Models","date":"2023-09-17","arxiv_id":"2309.09390","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-knowledge-distillation-via-flow","title":"Cross-lingual Knowledge Distillation via Flow-based Voice Conversion for Robust Polyglot Text-To-Speech","date":"2023-09-15","arxiv_id":"2309.08255","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompttts-controlling-speaker-identity-in","title":"PromptTTS++: Controlling Speaker Identity in Prompt-Based Text-to-Speech Using Natural Language Descriptions","date":"2023-09-15","arxiv_id":"2309.08140","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-text-to-speech-translation-system","title":"Direct Text to Speech Translation System using Acoustic Units","date":"2023-09-14","arxiv_id":"2309.07478","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-utterance-conditioned-vae-for-speech","title":"Cross-Utterance Conditioned VAE for Speech Generation","date":"2023-09-08","arxiv_id":"2309.04156","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-automatic-audiobook-creation","title":"Large-Scale Automatic Audiobook Creation","date":"2023-09-07","arxiv_id":"2309.03926","repositories_listed":0,"syntology":null},{"url":null,"slug":"grass-unified-generation-model-for-speech","title":"GRASS: Unified Generation Model for Speech-to-Semantic Tasks","date":"2023-09-06","arxiv_id":"2309.02780","repositories_listed":0,"syntology":null}],"record_sha256":"bf07ac2edab4bdb2e64f0421d2ca71d4f6ab10f83a7910464349e07d9c26fb8c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}