{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-synthesis/papers/2","list_of":"/task/text-to-speech-synthesis","task":"Text-To-Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":332,"counts":{"archive_papers_tagged":332,"with_a_code_link":104,"where_syntology_ran_a_sample":37,"not_listed_spam_title":0,"listed":332,"listed_where_code_ran":37,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":35,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":35,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-synthesis","prev":"/task/text-to-speech-synthesis","next":"/task/text-to-speech-synthesis/papers/3","papers":[{"url":"/paper/in-other-news-a-bi-style-text-to-speech-model","slug":"in-other-news-a-bi-style-text-to-speech-model","title":"In Other News: A Bi-style Text-to-speech Model for Synthesizing Newscaster Voice with Limited Data","date":"2019-04-04","arxiv_id":"1904.02790","repositories_listed":1,"syntology":null},{"url":"/paper/visualization-and-interpretation-of-latent","slug":"visualization-and-interpretation-of-latent","title":"Visualization and Interpretation of Latent Spaces for Controlling Expressive Speech Synthesis through Audio Analysis","date":"2019-03-27","arxiv_id":"1903.11570","repositories_listed":1,"syntology":null},{"url":"/paper/investigation-of-enhanced-tacotron-text-to","slug":"investigation-of-enhanced-tacotron-text-to","title":"Investigation of enhanced Tacotron text-to-speech synthesis systems with self-attention for pitch accent language","date":"2018-10-29","arxiv_id":"1810.11960","repositories_listed":1,"syntology":null},{"url":"/paper/the-emotional-voices-database-towards","slug":"the-emotional-voices-database-towards","title":"The Emotional Voices Database: Towards Controlling the Emotion Dimension in Voice Generation Systems","date":"2018-06-25","arxiv_id":"1806.09514","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/the-emotional-voices-database-towards#ran","syntology_url":"https://syntology.ai/paper/1806.09514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.09514"}},"official":{"repos":["numediart/EmoV-DB"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":null,"slug":"s2st-omni-an-efficient-and-scalable","title":"S2ST-Omni: An Efficient and Scalable Multilingual Speech-to-Speech Translation Framework via Seamless Speech-Text Alignment and Streaming Speech Generation","date":"2025-06-11","arxiv_id":"2506.11160","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-data-augmentation-approach-for","title":"A Novel Data Augmentation Approach for Automatic Speaking Assessment on Opinion Expressions","date":"2025-06-04","arxiv_id":"2506.04077","repositories_listed":0,"syntology":null},{"url":null,"slug":"capspeech-enabling-downstream-applications-in","title":"CapSpeech: Enabling Downstream Applications in Style-Captioned Text-to-Speech","date":"2025-06-03","arxiv_id":"2506.02863","repositories_listed":0,"syntology":null},{"url":null,"slug":"salf-mos-speaker-agnostic-latent-features","title":"SALF-MOS: Speaker Agnostic Latent Features Downsampled for MOS Prediction","date":"2025-06-02","arxiv_id":"2506.02082","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-thought-training-for-open-e2e-spoken","title":"Chain-of-Thought Training for Open E2E Spoken Dialogue Systems","date":"2025-05-31","arxiv_id":"2506.00722","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-streaming-text-to-speech-synthesis","title":"Zero-Shot Streaming Text to Speech Synthesis with Transducer and Auto-Regressive Modeling","date":"2025-05-26","arxiv_id":"2505.19669","repositories_listed":0,"syntology":null},{"url":null,"slug":"revival-with-voice-multi-modal-controllable","title":"Revival with Voice: Multi-modal Controllable Text-to-Speech Synthesis","date":"2025-05-25","arxiv_id":"2505.18972","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmsd-tts-few-shot-multi-speaker-multi-dialect","title":"FMSD-TTS: Few-shot Multi-Speaker Multi-Dialect Text-to-Speech Synthesis for Ü-Tsang, Amdo and Kham Speech Dataset Generation","date":"2025-05-20","arxiv_id":"2505.14351","repositories_listed":0,"syntology":null},{"url":null,"slug":"shallow-flow-matching-for-coarse-to-fine-text","title":"Shallow Flow Matching for Coarse-to-Fine Text-to-Speech Synthesis","date":"2025-05-18","arxiv_id":"2505.12226","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-end-to-end-text-to-speech","title":"Lightweight End-to-end Text-to-speech Synthesis for low resource on-device applications","date":"2025-05-12","arxiv_id":"2505.07701","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-framework-for-automated-1","title":"A Multi-Agent Framework for Automated Qinqiang Opera Script Generation Using Large Language Models","date":"2025-04-22","arxiv_id":"2504.15552","repositories_listed":0,"syntology":null},{"url":null,"slug":"autostyle-tts-retrieval-augmented-generation","title":"AutoStyle-TTS: Retrieval-Augmented Generation based Automatic Style Matching Text-to-Speech Synthesis","date":"2025-04-14","arxiv_id":"2504.10309","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-autoregressive-neural-codec-language","title":"Pseudo-Autoregressive Neural Codec Language Models for Efficient Zero-Shot Text-to-Speech Synthesis","date":"2025-04-14","arxiv_id":"2504.10352","repositories_listed":0,"syntology":null},{"url":null,"slug":"asvspoof-5-design-collection-and-validation","title":"ASVspoof 5: Design, Collection and Validation of Resources for Spoofing, Deepfake, and Adversarial Attack Detection Using Crowdsourced Speech","date":"2025-02-13","arxiv_id":"2502.08857","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-text-to-speech-synthesis-using","title":"Low-Resource Text-to-Speech Synthesis Using Noise-Augmented Training of ForwardTacotron","date":"2025-01-10","arxiv_id":"2501.05976","repositories_listed":0,"syntology":null},{"url":null,"slug":"proemo-prompt-driven-text-to-speech-synthesis","title":"PROEMO: Prompt-Driven Text-to-Speech Synthesis Based on Emotion and Intensity Control","date":"2025-01-10","arxiv_id":"2501.06276","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-speaker-specific-features-in-speaker","title":"Probing Speaker-specific Features in Speaker Representations","date":"2025-01-09","arxiv_id":"2501.05310","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-tts-stable-speaker-adaptive-text-to","title":"Stable-TTS: Stable Speaker-Adaptive Text-to-Speech Synthesis via Prosody Prompting","date":"2024-12-28","arxiv_id":"2412.20155","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-disentanglement-for-environment","title":"Incremental Disentanglement for Environment-Aware Zero-Shot Text-to-Speech Synthesis","date":"2024-12-22","arxiv_id":"2412.16977","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodyfm-unsupervised-phrasing-and","title":"ProsodyFM: Unsupervised Phrasing and Intonation Control for Intelligible Speech Synthesis","date":"2024-12-16","arxiv_id":"2412.11795","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-generative-modeling-with-residual","title":"Efficient Generative Modeling with Residual Vector Quantization-Based Tokens","date":"2024-12-13","arxiv_id":"2412.10208","repositories_listed":0,"syntology":null},{"url":null,"slug":"debatts-zero-shot-debating-text-to-speech","title":"Debatts: Zero-Shot Debating Text-to-Speech Synthesis","date":"2024-11-10","arxiv_id":"2411.06540","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-collecting-text-to","title":"A Unified Framework for Collecting Text-to-Speech Synthesis Datasets for 22 Indian Languages","date":"2024-10-18","arxiv_id":"2410.14197","repositories_listed":0,"syntology":null},{"url":null,"slug":"durian-e-2-duration-informed-attention","title":"DurIAN-E 2: Duration Informed Attention Network with Adaptive Variational Autoencoder and Adversarial Learning for Expressive Text-to-Speech Synthesis","date":"2024-10-17","arxiv_id":"2410.13288","repositories_listed":0,"syntology":null},{"url":null,"slug":"bahasa-harmony-a-comprehensive-dataset-for","title":"Bahasa Harmony: A Comprehensive Dataset for Bahasa Text-to-Speech Synthesis with Discrete Codec Modeling of EnGen-TTS","date":"2024-10-09","arxiv_id":"2410.06608","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-training-strategies-for-natural","title":"Efficient training strategies for natural sounding speech synthesis and speaker adaptation based on FastPitch","date":"2024-10-09","arxiv_id":"2410.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"hall-e-hierarchical-neural-codec-language","title":"HALL-E: Hierarchical Neural Codec Language Model for Minute-Long Zero-Shot Text-to-Speech Synthesis","date":"2024-10-06","arxiv_id":"2410.04380","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-semantic-communication-for-text-to","title":"Generative Semantic Communication for Text-to-Speech Synthesis","date":"2024-10-04","arxiv_id":"2410.03459","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-conversion-using-discrete-units-with","title":"Accent conversion using discrete units with parallel data synthesized from controllable accented TTS","date":"2024-09-30","arxiv_id":"2410.03734","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylefusion-tts-multimodal-style-control-and","title":"StyleFusion TTS: Multimodal Style-control and Enhanced Feature Fusion for Zero-shot Text-to-speech Synthesis","date":"2024-09-24","arxiv_id":"2409.15741","repositories_listed":0,"syntology":null},{"url":null,"slug":"styletts-zs-efficient-high-quality-zero-shot","title":"StyleTTS-ZS: Efficient High-Quality Zero-Shot Text-to-Speech Synthesis with Distilled Time-Varying Style Diffusion","date":"2024-09-16","arxiv_id":"2409.10058","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-in-the-wild","title":"Text-To-Speech Synthesis In The Wild","date":"2024-09-13","arxiv_id":"2409.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-text-error-correction-for-chinese-speech","title":"Full-text Error Correction for Chinese Speech Recognition with Large Language Model","date":"2024-09-12","arxiv_id":"2409.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-happens-to-diffusion-model-likelihood","title":"What happens to diffusion model likelihood when your model is conditional?","date":"2024-09-10","arxiv_id":"2409.06364","repositories_listed":0,"syntology":null},{"url":null,"slug":"as-speech-adaptive-style-for-speech-synthesis","title":"AS-Speech: Adaptive Style For Speech Synthesis","date":"2024-09-09","arxiv_id":"2409.05730","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-bandwidth-expansion-via-high-fidelity","title":"Speech Bandwidth Expansion Via High Fidelity Generative Adversarial Networks","date":"2024-07-26","arxiv_id":"2407.18571","repositories_listed":0,"syntology":null},{"url":null,"slug":"spontaneous-style-text-to-speech-synthesis","title":"Spontaneous Style Text-to-Speech Synthesis with Controllable Spontaneous Behaviors Based on Language Models","date":"2024-07-18","arxiv_id":"2407.13509","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-speech-synthesis-without","title":"Autoregressive Speech Synthesis without Vector Quantization","date":"2024-07-11","arxiv_id":"2407.08551","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accented-speech-recognition-using","title":"Improving Accented Speech Recognition using Data Augmentation based on Unsupervised Text-to-Speech Synthesis","date":"2024-07-04","arxiv_id":"2407.04047","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-zero-shot-text-to-speech-synthesis","title":"Robust Zero-Shot Text-to-Speech Synthesis with Reverse Inference Optimization","date":"2024-07-02","arxiv_id":"2407.02243","repositories_listed":0,"syntology":null},{"url":null,"slug":"fly-tts-fast-lightweight-and-high-quality-end","title":"FLY-TTS: Fast, Lightweight and High-Quality End-to-End Text-to-Speech Synthesis","date":"2024-06-30","arxiv_id":"2407.00753","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-accent-modeling-with","title":"Multi-Scale Accent Modeling and Disentangling for Multi-Speaker Multi-Accent Text-to-Speech Synthesis","date":"2024-06-16","arxiv_id":"2406.10844","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-e-r-robust-and-efficient-zero-shot-text","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","date":"2024-06-12","arxiv_id":"2406.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-diffusion-transformer-for-text","title":"Autoregressive Diffusion Transformer for Text-to-Speech Synthesis","date":"2024-06-08","arxiv_id":"2406.05551","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-e-2-neural-codec-language-models-are","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","date":"2024-06-08","arxiv_id":"2406.05370","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-audio-codec-based-zero-shot-text-to","title":"Improving Audio Codec-based Zero-Shot Text-to-Speech Synthesis with Multi-Modal Context and Large Language Model","date":"2024-06-06","arxiv_id":"2406.03706","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-mixture-of-experts-for-expressive-text","title":"Style Mixture of Experts for Expressive Text-To-Speech Synthesis","date":"2024-06-05","arxiv_id":"2406.03637","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-enhanced-language-modeling-for-text","title":"Phonetic Enhanced Language Modeling for Text-to-Speech Synthesis","date":"2024-06-04","arxiv_id":"2406.02009","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-zero-shot-text-to-speech-synthesis","title":"Enhancing Zero-shot Text-to-Speech Synthesis with Human Feedback","date":"2024-06-02","arxiv_id":"2406.00654","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-fine-tuning-text-1","title":"DLPO: Diffusion Model Loss-Guided Reinforcement Learning for Fine-Tuning Text-to-Speech Diffusion Models","date":"2024-05-23","arxiv_id":"2405.14632","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-text-to-speech-synthesis-from-a","title":"Evaluating Text-to-Speech Synthesis from a Large Discrete Token-based Speech Language Model","date":"2024-05-16","arxiv_id":"2405.09768","repositories_listed":0,"syntology":null},{"url":null,"slug":"rall-e-robust-codec-language-modeling-with","title":"RALL-E: Robust Codec Language Modeling with Chain-of-Thought Prompting for Text-to-Speech Synthesis","date":"2024-04-04","arxiv_id":"2404.03204","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptcodec-high-fidelity-neural-speech-codec","title":"PSCodec: A Series of High-Fidelity Low-bitrate Neural Speech Codecs Leveraging Prompt Encoders","date":"2024-04-03","arxiv_id":"2404.02702","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-zero-shot-text-to-speech","title":"Noise-robust zero-shot text-to-speech synthesis conditioned on self-supervised speech-representation model with adapters","date":"2024-01-10","arxiv_id":"2401.05111","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-large-language-model-for-speech","title":"Boosting Large Language Model for Speech Synthesis: An Empirical Study","date":"2023-12-30","arxiv_id":"2401.00246","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalization-of-lithuanian-text-using","title":"Normalization of Lithuanian Text Using Regular Expressions","date":"2023-12-29","arxiv_id":"2312.17660","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-tts-multi-modal-prompt-based-style","title":"MM-TTS: Multi-modal Prompt based Style Transfer for Expressive Text-to-Speech Synthesis","date":"2023-12-17","arxiv_id":"2312.10687","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-study-assessing-the-combined","title":"An Experimental Study: Assessing the Combined Framework of WavLM and BEST-RQ for Text-to-Speech Synthesis","date":"2023-12-08","arxiv_id":"2312.05415","repositories_listed":0,"syntology":null},{"url":null,"slug":"schrodinger-bridges-beat-diffusion-models-on","title":"Schrodinger Bridges Beat Diffusion Models on Text-to-Speech Synthesis","date":"2023-12-06","arxiv_id":"2312.03491","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-mixed-text-to-speech-synthesis-under-low","title":"Code-Mixed Text to Speech Synthesis under Low-Resource Constraints","date":"2023-12-02","arxiv_id":"2312.01103","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-flows-for-generative-modeling-and","title":"Guided Flows for Generative Modeling and Decision Making","date":"2023-11-22","arxiv_id":"2311.13443","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-pre-training-for-speech-with-flow","title":"Generative Pre-training for Speech with Flow Matching","date":"2023-10-25","arxiv_id":"2310.16338","repositories_listed":0,"syntology":null},{"url":"/paper/unified-speech-and-gesture-synthesis-using","slug":"unified-speech-and-gesture-synthesis-using","title":"Unified speech and gesture synthesis using flow matching","date":"2023-10-08","arxiv_id":"2310.05181","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-voicemos-challenge-2023-zero-shot","title":"The VoiceMOS Challenge 2023: Zero-shot Subjective Speech Quality Prediction for Multiple Domains","date":"2023-10-04","arxiv_id":"2310.02640","repositories_listed":0,"syntology":null},{"url":null,"slug":"durian-e-duration-informed-attention-network","title":"DurIAN-E: Duration Informed Attention Network For Expressive Text-to-Speech Synthesis","date":"2023-09-22","arxiv_id":"2309.12792","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-fruitshell-french-synthesis-system-at-the","title":"The FruitShell French synthesis system at the Blizzard 2023 Challenge","date":"2023-09-01","arxiv_id":"2309.00223","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spontaneous-style-modeling-with-semi","title":"Towards Spontaneous Style Modeling with Semi-supervised Pre-training for Conversational Text-to-Speech Synthesis","date":"2023-08-31","arxiv_id":"2308.16593","repositories_listed":0,"syntology":null},{"url":null,"slug":"saltts-leveraging-self-supervised-speech","title":"SALTTS: Leveraging Self-Supervised Speech Representations for improved Text-to-Speech Synthesis","date":"2023-08-02","arxiv_id":"2308.01018","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-normalizing-flows-and-diffusion","title":"Comparing normalizing flows and diffusion models for prosody and acoustic modelling in text-to-speech","date":"2023-07-31","arxiv_id":"2307.16679","repositories_listed":0,"syntology":null},{"url":null,"slug":"slmgan-exploiting-speech-language-model","title":"SLMGAN: Exploiting Speech Language Model Representations for Unsupervised Zero-Shot Voice Conversion in GANs","date":"2023-07-18","arxiv_id":"2307.09435","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-automatic-voice-over-with","title":"High-Quality Automatic Voice Over with Accurate Alignment: Supervision through Self-Supervised Discrete Speech Units","date":"2023-06-29","arxiv_id":"2306.17005","repositories_listed":0,"syntology":null},{"url":null,"slug":"zet-speech-zero-shot-adaptive-emotion","title":"ZET-Speech: Zero-shot adaptive Emotion-controllable Text-to-Speech Synthesis with Diffusion and Style-based Models","date":"2023-05-23","arxiv_id":"2305.13831","repositories_listed":0,"syntology":null},{"url":null,"slug":"vakta-setu-a-speech-to-speech-machine","title":"VAKTA-SETU: A Speech-to-Speech Machine Translation Service in Select Indic Languages","date":"2023-05-21","arxiv_id":"2305.12518","repositories_listed":0,"syntology":null},{"url":null,"slug":"mparrottts-multilingual-multi-speaker-text-to","title":"MParrotTTS: Multilingual Multi-speaker Text to Speech Synthesis in Low Resource Setting","date":"2023-05-19","arxiv_id":"2305.11926","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-front-end-framework-for-english","title":"A unified front-end framework for English text-to-speech synthesis","date":"2023-05-18","arxiv_id":"2305.10666","repositories_listed":0,"syntology":null},{"url":null,"slug":"accented-text-to-speech-synthesis-with","title":"Accented Text-to-Speech Synthesis with Limited Data","date":"2023-05-08","arxiv_id":"2305.04816","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2-ctts-end-to-end-multi-scale-multi-modal","title":"M2-CTTS: End-to-End Multi-scale Multi-modal Conversational Text-to-Speech Synthesis","date":"2023-05-03","arxiv_id":"2305.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-learning-techniques-for-3","title":"A Review of Deep Learning Techniques for Speech Processing","date":"2023-04-30","arxiv_id":"2305.00359","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-synthesis","title":"Zero-shot text-to-speech synthesis conditioned using self-supervised speech representation model","date":"2023-04-24","arxiv_id":"2304.11976","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-is-all-you-need-personalizing-asr-models","title":"Text is All You Need: Personalizing ASR Models using Controllable Speech Synthesis","date":"2023-03-27","arxiv_id":"2303.14885","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-diffusion-model-for-speech-synthesis-a","title":"A Survey on Audio Diffusion Models: Text To Speech Synthesis and Enhancement in Generative AI","date":"2023-03-23","arxiv_id":"2303.13336","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-high-dimensional-data-with-sparse","title":"Controllable Prosody Generation With Partial Inputs","date":"2023-03-14","arxiv_id":"2303.09446","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-prosody-transfer-models-transfer-prosody","title":"Do Prosody Transfer Models Transfer Prosody?","date":"2023-03-07","arxiv_id":"2303.04289","repositories_listed":0,"syntology":null},{"url":null,"slug":"parrottts-text-to-speech-synthesis-by","title":"ParrotTTS: Text-to-Speech synthesis by exploiting self-supervised representations","date":"2023-03-01","arxiv_id":"2303.01261","repositories_listed":0,"syntology":null},{"url":null,"slug":"uzbektagger-the-rule-based-pos-tagger-for","title":"UzbekTagger: The rule-based POS tagger for Uzbek language","date":"2023-01-30","arxiv_id":"2301.12711","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-automated-machine-translation-to","title":"Applying Automated Machine Translation to Educational Video Courses","date":"2023-01-09","arxiv_id":"2301.03141","repositories_listed":0,"syntology":null},{"url":null,"slug":"revise-self-supervised-speech-resynthesis-1","title":"ReVISE: Self-Supervised Speech Resynthesis With Visual Input for Universal and Generalized Speech Regeneration","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/revise-self-supervised-speech-resynthesis","slug":"revise-self-supervised-speech-resynthesis","title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","date":"2022-12-21","arxiv_id":"2212.11377","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-japanese-png-bert-language","title":"Investigation of Japanese PnG BERT language model in text-to-speech synthesis for pitch accent language","date":"2022-12-16","arxiv_id":"2212.08321","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-based-on-latent","title":"Text-to-speech synthesis based on latent variable conversion using diffusion probabilistic model and variational autoencoder","date":"2022-12-16","arxiv_id":"2212.08329","repositories_listed":0,"syntology":null},{"url":null,"slug":"any-speaker-adaptive-text-to-speech-synthesis","title":"Grad-StyleSpeech: Any-speaker Adaptive Text-to-Speech Synthesis with Diffusion Models","date":"2022-11-17","arxiv_id":"2211.09383","repositories_listed":0,"syntology":null},{"url":null,"slug":"technology-pipeline-for-large-scale-cross","title":"Technology Pipeline for Large Scale Cross-Lingual Dubbing of Lecture Videos into Multiple Indian Languages","date":"2022-11-01","arxiv_id":"2211.01338","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtuoso-massive-multilingual-speech-text","title":"Virtuoso: Massive Multilingual Speech-Text Joint Semi-Supervised Learning for Text-To-Speech","date":"2022-10-27","arxiv_id":"2210.15447","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-affective-speech-synthesis-and","title":"An Overview of Affective Speech Synthesis and Conversion in the Deep Learning Era","date":"2022-10-06","arxiv_id":"2210.03538","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-accented-text-to-speech","title":"Controllable Accented Text-to-Speech Synthesis","date":"2022-09-22","arxiv_id":"2209.10804","repositories_listed":0,"syntology":null},{"url":null,"slug":"epic-tts-models-empirical-pruning","title":"EPIC TTS Models: Empirical Pruning Investigations Characterizing Text-To-Speech Models","date":"2022-09-22","arxiv_id":"2209.10890","repositories_listed":0,"syntology":null}],"record_sha256":"b2537a5487d5e62544952ef71c33e13278c39b5d3aab9a567fb1fe0195cd6828","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}