{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/6","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":13,"rows_per_page":100,"rows":[501,600],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/5","next":"/task/speech-synthesis/papers/7","papers":[{"url":null,"slug":"which-prosodic-features-matter-most-for","title":"Which Prosodic Features Matter Most for Pragmatics?","date":"2024-08-23","arxiv_id":"2408.13240","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-based-ivr","title":"AI-Based IVR","date":"2024-08-20","arxiv_id":"2408.10549","repositories_listed":0,"syntology":null},{"url":null,"slug":"vnet-a-gan-based-multi-tier-discriminator","title":"VNet: A GAN-based Multi-Tier Discriminator Network for Speech Synthesis Vocoders","date":"2024-08-13","arxiv_id":"2408.06906","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00284","title":"Bailing-TTS: Chinese Dialectal Speech Synthesis Towards Human-like Spontaneous Representation","date":"2024-08-01","arxiv_id":"2408.00284","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-bandwidth-expansion-via-high-fidelity","title":"Speech Bandwidth Expansion Via High Fidelity Generative Adversarial Networks","date":"2024-07-26","arxiv_id":"2407.18571","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-improving-nam-to-speech-synthesis","title":"Towards Improving NAM-to-Speech Synthesis Intelligibility using Self-Supervised Speech Models","date":"2024-07-26","arxiv_id":"2407.18541","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-speaker-modeling-and-its","title":"Overview of Speaker Modeling and Its Applications: From the Lens of Deep Speaker Representation Learning","date":"2024-07-21","arxiv_id":"2407.15188","repositories_listed":0,"syntology":null},{"url":null,"slug":"mscenespeech-a-multi-scene-speech-dataset-for","title":"MSceneSpeech: A Multi-Scene Speech Dataset For Expressive Speech Synthesis","date":"2024-07-19","arxiv_id":"2407.14006","repositories_listed":0,"syntology":null},{"url":null,"slug":"spontaneous-style-text-to-speech-synthesis","title":"Spontaneous Style Text-to-Speech Synthesis with Controllable Spontaneous Behaviors Based on Language Models","date":"2024-07-18","arxiv_id":"2407.13509","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-speech-synthesis-without","title":"Autoregressive Speech Synthesis without Vector Quantization","date":"2024-07-11","arxiv_id":"2407.08551","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-accessible-comics-for-blind-and-low","title":"Toward accessible comics for blind and low vision readers","date":"2024-07-11","arxiv_id":"2407.08248","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-speech-unit-selection-for-textless","title":"Analyzing Speech Unit Selection for Textless Speech-to-Speech Translation","date":"2024-07-08","arxiv_id":"2407.18332","repositories_listed":0,"syntology":null},{"url":null,"slug":"fa-gan-artifacts-free-and-phase-aware-high","title":"FA-GAN: Artifacts-free and Phase-aware High-fidelity GAN-based Vocoder","date":"2024-07-05","arxiv_id":"2407.04575","repositories_listed":0,"syntology":null},{"url":null,"slug":"we-need-variations-in-speech-synthesis-sub","title":"We Need Variations in Speech Generation: Sub-center Modelling for Speaker Embeddings","date":"2024-07-05","arxiv_id":"2407.04291","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accented-speech-recognition-using","title":"Improving Accented Speech Recognition using Data Augmentation based on Unsupervised Text-to-Speech Synthesis","date":"2024-07-04","arxiv_id":"2407.04047","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-zero-shot-text-to-speech-synthesis","title":"Robust Zero-Shot Text-to-Speech Synthesis with Reverse Inference Optimization","date":"2024-07-02","arxiv_id":"2407.02243","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-diffusion-models","title":"A Comprehensive Survey on Diffusion Models and Their Applications","date":"2024-07-01","arxiv_id":"2408.10207","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-zero-shot-text-to-speech-with","title":"Lightweight Zero-shot Text-to-Speech with Mixture of Adapters","date":"2024-07-01","arxiv_id":"2407.01291","repositories_listed":0,"syntology":null},{"url":null,"slug":"fly-tts-fast-lightweight-and-high-quality-end","title":"FLY-TTS: Fast, Lightweight and High-Quality End-to-End Text-to-Speech Synthesis","date":"2024-06-30","arxiv_id":"2407.00753","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-fidelity-text-to-speech-via-discrete","title":"High Fidelity Text-to-Speech Via Discrete Tokens Using Token Transducer and Group Masked Language Model","date":"2024-06-25","arxiv_id":"2406.17310","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-robustness-of-llm-based-speech","title":"Improving Robustness of LLM-based Speech Synthesis by Learning Monotonic Alignment","date":"2024-06-25","arxiv_id":"2406.17957","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-parameter-efficient-transfer","title":"Leveraging Parameter-Efficient Transfer Learning for Multi-Lingual Text-to-Speech Adaptation","date":"2024-06-25","arxiv_id":"2406.17257","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-text-to-speech-for-arabic","title":"Towards Zero-Shot Text-To-Speech for Arabic Dialects","date":"2024-06-24","arxiv_id":"2406.16751","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-speaker-multi-lingual-voice-cloning","title":"A multi-speaker multi-lingual voice cloning system based on vits2 for limmits 2024 challenge","date":"2024-06-22","arxiv_id":"2406.17801","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mel-spectrogram-enhancement-paradigm-based","title":"A Mel Spectrogram Enhancement Paradigm Based on CWT in Speech Synthesis","date":"2024-06-18","arxiv_id":"2406.12164","repositories_listed":0,"syntology":null},{"url":null,"slug":"1000-african-voices-advancing-inclusive-multi","title":"1000 African Voices: Advancing inclusive multi-speaker multi-accent speech synthesis","date":"2024-06-17","arxiv_id":"2406.11727","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-accent-modeling-with","title":"Multi-Scale Accent Modeling and Disentangling for Multi-Speaker Multi-Accent Text-to-Speech Synthesis","date":"2024-06-16","arxiv_id":"2406.10844","repositories_listed":0,"syntology":null},{"url":null,"slug":"toneunit-a-speech-discretization-approach-for","title":"ToneUnit: A Speech Discretization Approach for Tonal Language Speech Synthesis","date":"2024-06-13","arxiv_id":"2406.08989","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyspeech-exploring-unified-multitask-speech","title":"PolySpeech: Exploring Unified Multitask Speech Models for Competitiveness with Single-task Models","date":"2024-06-12","arxiv_id":"2406.07801","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-e-r-robust-and-efficient-zero-shot-text","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","date":"2024-06-12","arxiv_id":"2406.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-achieve-high-quality-direct-speech-to","title":"Can We Achieve High-quality Direct Speech-to-Speech Translation without Parallel Speech Data?","date":"2024-06-11","arxiv_id":"2406.07289","repositories_listed":0,"syntology":null},{"url":null,"slug":"codecfake-enhancing-anti-spoofing-models","title":"CodecFake: Enhancing Anti-Spoofing Models Against Deepfake Audios from Codec-Based Speech Synthesis Systems","date":"2024-06-11","arxiv_id":"2406.07237","repositories_listed":0,"syntology":null},{"url":null,"slug":"jengan-stacked-shifted-filters-in-gan-based","title":"JenGAN: Stacked Shifted Filters in GAN-Based Speech Synthesis","date":"2024-06-10","arxiv_id":"2406.06111","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-aware-and-context-aware-expressive","title":"Text-aware and Context-aware Expressive Audiobook Speech Synthesis","date":"2024-06-09","arxiv_id":"2406.05672","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-diffusion-transformer-for-text","title":"Autoregressive Diffusion Transformer for Text-to-Speech Synthesis","date":"2024-06-08","arxiv_id":"2406.05551","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-e-2-neural-codec-language-models-are","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","date":"2024-06-08","arxiv_id":"2406.05370","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-codecs-spectrogram-based-audio","title":"Spectral Codecs: Improving Non-Autoregressive Speech Synthesis with Spectrogram-Based Audio Codecs","date":"2024-06-07","arxiv_id":"2406.05298","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-audio-codec-based-zero-shot-text-to","title":"Improving Audio Codec-based Zero-Shot Text-to-Speech Synthesis with Multi-Modal Context and Large Language Model","date":"2024-06-06","arxiv_id":"2406.03706","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-mixture-of-experts-for-expressive-text","title":"Style Mixture of Experts for Expressive Text-To-Speech Synthesis","date":"2024-06-05","arxiv_id":"2406.03637","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-enhanced-language-modeling-for-text","title":"Phonetic Enhanced Language Modeling for Text-to-Speech Synthesis","date":"2024-06-04","arxiv_id":"2406.02009","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-conversion-in-text-to-speech-using","title":"Accent Conversion in Text-To-Speech Using Multi-Level VAE and Adversarial Training","date":"2024-06-03","arxiv_id":"2406.01018","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-zero-shot-text-to-speech-synthesis","title":"Enhancing Zero-shot Text-to-Speech Synthesis with Human Feedback","date":"2024-06-02","arxiv_id":"2406.00654","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-prosody-transfer-comparing","title":"Multilingual Prosody Transfer: Comparing Supervised & Transfer Learning","date":"2024-05-23","arxiv_id":"2406.00022","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-fine-tuning-text-1","title":"DLPO: Diffusion Model Loss-Guided Reinforcement Learning for Fine-Tuning Text-to-Speech Diffusion Models","date":"2024-05-23","arxiv_id":"2405.14632","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-text-to-speech-synthesis-from-a","title":"Evaluating Text-to-Speech Synthesis from a Large Discrete Token-based Speech Language Model","date":"2024-05-16","arxiv_id":"2405.09768","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressivity-and-speech-synthesis","title":"Expressivity and Speech Synthesis","date":"2024-04-30","arxiv_id":"2404.19363","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-audio-deepfake-detection","title":"Retrieval-Augmented Audio Deepfake Detection","date":"2024-04-22","arxiv_id":"2404.13892","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-efficient-fine-tuning-a","title":"Parameter Efficient Fine Tuning: A Comprehensive Analysis Across Applications","date":"2024-04-21","arxiv_id":"2404.13506","repositories_listed":0,"syntology":null},{"url":null,"slug":"rall-e-robust-codec-language-modeling-with","title":"RALL-E: Robust Codec Language Modeling with Chain-of-Thought Prompting for Text-to-Speech Synthesis","date":"2024-04-04","arxiv_id":"2404.03204","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-the-interplay-between-syntactic","title":"Leveraging the Interplay Between Syntactic and Acoustic Cues for Optimizing Korean TTS Pause Formation","date":"2024-04-03","arxiv_id":"2404.02592","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptcodec-high-fidelity-neural-speech-codec","title":"PSCodec: A Series of High-Fidelity Low-bitrate Neural Speech Codecs Leveraging Prompt Encoders","date":"2024-04-03","arxiv_id":"2404.02702","repositories_listed":0,"syntology":null},{"url":null,"slug":"removing-speaker-information-from-speech","title":"Removing Speaker Information from Speech Representation using Variable-Length Soft Pooling","date":"2024-04-01","arxiv_id":"2404.00856","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-generative-adversarial-network-based","title":"Training Generative Adversarial Network-Based Vocoder with Limited Data Using Augmentation-Conditional Discriminator","date":"2024-03-25","arxiv_id":"2403.16464","repositories_listed":0,"syntology":null},{"url":"/paper/m-3-av-a-multimodal-multigenre-and","slug":"m-3-av-a-multimodal-multigenre-and","title":"M$^3$AV: A Multimodal, Multigenre, and Multipurpose Audio-Visual Academic Lecture Dataset","date":"2024-03-21","arxiv_id":"2403.14168","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-speech-language-models","title":"An Empirical Study of Speech Language Models for Prompt-Conditioned Speech Synthesis","date":"2024-03-19","arxiv_id":"2403.12402","repositories_listed":0,"syntology":null},{"url":null,"slug":"em-tts-efficiently-trained-low-resource","title":"EM-TTS: Efficiently Trained Low-Resource Mongolian Lightweight Text-to-Speech","date":"2024-03-13","arxiv_id":"2403.08164","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-accurate-lip-to-speech-synthesis-in","title":"Towards Accurate Lip-to-Speech Synthesis in-the-Wild","date":"2024-03-02","arxiv_id":"2403.01087","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxgenesis-unsupervised-discovery-of-latent","title":"VoxGenesis: Unsupervised Discovery of Latent Speaker Manifold for Speech Synthesis","date":"2024-03-01","arxiv_id":"2403.00529","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-multilingual-speech-synthesis-to","title":"Extending Multilingual Speech Synthesis to 100+ Languages without Transcribed Data","date":"2024-02-29","arxiv_id":"2402.18932","repositories_listed":0,"syntology":null},{"url":"/paper/speaking-in-wavelet-domain-a-simple-and","slug":"speaking-in-wavelet-domain-a-simple-and","title":"Speaking in Wavelet Domain: A Simple and Efficient Approach to Speed up Speech Diffusion Model","date":"2024-02-16","arxiv_id":"2402.10642","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/speaking-in-wavelet-domain-a-simple-and#ran","syntology_url":"https://syntology.ai/paper/2402.10642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10642"}},"official":null}},{"url":null,"slug":"speech-rhythm-based-speaker-embeddings","title":"Speech Rhythm-Based Speaker Embeddings Extraction from Phonemes and Phoneme Duration for Multi-Speaker Speech Synthesis","date":"2024-02-11","arxiv_id":"2402.07085","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechcomposer-unifying-multiple-speech-tasks","title":"SpeechComposer: Unifying Multiple Speech Tasks with Prompt Composition","date":"2024-01-31","arxiv_id":"2401.18045","repositories_listed":0,"syntology":null},{"url":null,"slug":"specdiff-gan-a-spectrally-shaped-noise","title":"SpecDiff-GAN: A Spectrally-Shaped Noise Diffusion GAN for Speech and Music Synthesis","date":"2024-01-30","arxiv_id":"2402.01753","repositories_listed":0,"syntology":null},{"url":null,"slug":"muntts-a-text-to-speech-system-for-mundari","title":"MunTTS: A Text-to-Speech System for Mundari","date":"2024-01-28","arxiv_id":"2401.15579","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-accessibility-voice-cloning-and","title":"Empowering Communication: Speech Technology for Indian and Western Accents through AI-powered Speech Synthesis","date":"2024-01-22","arxiv_id":"2401.11771","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultra-lightweight-neural-differential-dsp","title":"Ultra-lightweight Neural Differential DSP Vocoder For High Quality Speech Synthesis","date":"2024-01-19","arxiv_id":"2401.10460","repositories_listed":0,"syntology":null},{"url":null,"slug":"ed-tts-multi-scale-emotion-modeling-using","title":"ED-TTS: Multi-Scale Emotion Modeling using Cross-Domain Emotion Diarization for Emotional Speech Synthesis","date":"2024-01-16","arxiv_id":"2401.08166","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-zero-shot-text-to-speech","title":"Noise-robust zero-shot text-to-speech synthesis conditioned on self-supervised speech-representation model with adapters","date":"2024-01-10","arxiv_id":"2401.05111","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamvc-real-time-low-latency-voice","title":"StreamVC: Real-Time Low-Latency Voice Conversion","date":"2024-01-05","arxiv_id":"2401.03078","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-fastpitch-chunk-based-high","title":"Incremental FastPitch: Chunk-based High Quality Text to Speech","date":"2024-01-03","arxiv_id":"2401.01755","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-large-language-model-for-speech","title":"Boosting Large Language Model for Speech Synthesis: An Empirical Study","date":"2023-12-30","arxiv_id":"2401.00246","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalization-of-lithuanian-text-using","title":"Normalization of Lithuanian Text Using Regular Expressions","date":"2023-12-29","arxiv_id":"2312.17660","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-new-voices-using-normalizing-flows","title":"Creating New Voices using Normalizing Flows","date":"2023-12-22","arxiv_id":"2312.14569","repositories_listed":0,"syntology":null},{"url":null,"slug":"braintalker-low-resource-brain-to-speech","title":"BrainTalker: Low-Resource Brain-to-Speech Synthesis with Transfer Learning using Wav2Vec 2.0","date":"2023-12-21","arxiv_id":"2312.13600","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-speech-in-speech-perception-via-a","title":"Evaluating Speech-in-Speech Perception via a Humanoid Robot","date":"2023-12-19","arxiv_id":"2312.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylespeech-self-supervised-style-enhancing","title":"StyleSpeech: Self-supervised Style Enhancing with VQ-VAE-based Pre-training for Expressive Audiobook Speech Synthesis","date":"2023-12-19","arxiv_id":"2312.12181","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-tts-multi-modal-prompt-based-style","title":"MM-TTS: Multi-modal Prompt based Style Transfer for Expressive Text-to-Speech Synthesis","date":"2023-12-17","arxiv_id":"2312.10687","repositories_listed":0,"syntology":null},{"url":null,"slug":"concss-contrastive-based-context","title":"CONCSS: Contrastive-based Context Comprehension for Dialogue-appropriate Prosody in Conversational Speech Synthesis","date":"2023-12-16","arxiv_id":"2312.10358","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-speech-embeddings-for-speech-synthesis","title":"Neural Speech Embeddings for Speech Synthesis Based on Deep Generative Networks","date":"2023-12-10","arxiv_id":"2312.05814","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-study-assessing-the-combined","title":"An Experimental Study: Assessing the Combined Framework of WavLM and BEST-RQ for Text-to-Speech Synthesis","date":"2023-12-08","arxiv_id":"2312.05415","repositories_listed":0,"syntology":null},{"url":null,"slug":"schrodinger-bridges-beat-diffusion-models-on","title":"Schrodinger Bridges Beat Diffusion Models on Text-to-Speech Synthesis","date":"2023-12-06","arxiv_id":"2312.03491","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-mixed-text-to-speech-synthesis-under-low","title":"Code-Mixed Text to Speech Synthesis under Low-Resource Constraints","date":"2023-12-02","arxiv_id":"2312.01103","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-flows-for-generative-modeling-and","title":"Guided Flows for Generative Modeling and Decision Making","date":"2023-11-22","arxiv_id":"2311.13443","repositories_listed":0,"syntology":null},{"url":null,"slug":"encoding-speaker-specific-latent-speech","title":"ELF: Encoding Speaker-Specific Latent Speech Feature for Speech Synthesis","date":"2023-11-20","arxiv_id":"2311.11745","repositories_listed":0,"syntology":null},{"url":null,"slug":"le-ssl-mos-self-supervised-learning-mos","title":"LE-SSL-MOS: Self-Supervised Learning MOS Prediction with Listener Enhancement","date":"2023-11-17","arxiv_id":"2311.10656","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-opportunities-of-green-computing-a","title":"On the Opportunities of Green Computing: A Survey","date":"2023-11-01","arxiv_id":"2311.00447","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-generation-of-artificial-speaker","title":"Controllable Generation of Artificial Speaker Embeddings through Discovery of Principal Directions","date":"2023-10-26","arxiv_id":"2310.17502","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-expressive-speech-synthesis-via-1","title":"Boosting Multi-Speaker Expressive Speech Synthesis with Semi-supervised Contrastive Learning","date":"2023-10-26","arxiv_id":"2310.17101","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-pre-training-for-speech-with-flow","title":"Generative Pre-training for Speech with Flow Matching","date":"2023-10-25","arxiv_id":"2310.16338","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-based-models-for-speech-synthesis","title":"Energy-Based Models For Speech Synthesis","date":"2023-10-19","arxiv_id":"2310.12765","repositories_listed":0,"syntology":null},{"url":null,"slug":"selfvc-voice-conversion-with-iterative","title":"SelfVC: Voice Conversion With Iterative Refinement using Self Transformations","date":"2023-10-14","arxiv_id":"2310.09653","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaking-rate-attention-based-duration","title":"Speaking rate attention-based duration prediction for speed control TTS","date":"2023-10-13","arxiv_id":"2310.08846","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-oriented-manipulation-of-speaker","title":"Privacy-oriented manipulation of speaker representations","date":"2023-10-10","arxiv_id":"2310.06652","repositories_listed":0,"syntology":null},{"url":"/paper/neutral-tts-female-voice-corpus-in-brazilian","slug":"neutral-tts-female-voice-corpus-in-brazilian","title":"Neutral TTS Female Voice Corpus in Brazilian Portuguese","date":"2023-10-08","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-filling-latent-space-data-augmentation","title":"Latent Filling: Latent Space Data Augmentation for Zero-shot Speech Synthesis","date":"2023-10-05","arxiv_id":"2310.03538","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-voicemos-challenge-2023-zero-shot","title":"The VoiceMOS Challenge 2023: Zero-shot Subjective Speech Quality Prediction for Multiple Domains","date":"2023-10-04","arxiv_id":"2310.02640","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-fidelity-speech-synthesis-with-minimal","title":"High-Fidelity Speech Synthesis with Minimal Supervision: All Using Diffusion Models","date":"2023-09-27","arxiv_id":"2309.15512","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-watermarking-for-adversarial","title":"Collaborative Watermarking for Adversarial Speech Synthesis","date":"2023-09-26","arxiv_id":"2309.15224","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-stylespeech-improved-face-to-voice","title":"Face-StyleSpeech: Enhancing Zero-shot Speech Synthesis from Face Images with Improved Face-to-Speech Mapping","date":"2023-09-25","arxiv_id":"2311.05844","repositories_listed":0,"syntology":null},{"url":null,"slug":"durian-e-duration-informed-attention-network","title":"DurIAN-E: Duration Informed Attention Network For Expressive Text-to-Speech Synthesis","date":"2023-09-22","arxiv_id":"2309.12792","repositories_listed":0,"syntology":null}],"record_sha256":"5a0352dfb2b2d72cb8d0efd1d059a6eed7c118585cf10a80520463b1191a4efa","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}