{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/7","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":13,"rows_per_page":100,"rows":[601,700],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/6","next":"/task/speech-synthesis/papers/8","papers":[{"url":null,"slug":"a-discourse-level-multi-scale-prosodic-model","title":"A Discourse-level Multi-scale Prosodic Model for Fine-grained Emotion Analysis","date":"2023-09-21","arxiv_id":"2309.11849","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-while-you-think-streaming-speech","title":"Speak While You Think: Streaming Speech Synthesis During Text Generation","date":"2023-09-20","arxiv_id":"2309.11210","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speech-enhancement-for-low-resource","title":"Exploring Speech Enhancement for Low-resource Speech Synthesis","date":"2023-09-19","arxiv_id":"2309.10795","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-speech-ptm-text-llm-and-emotional","title":"Leveraging Speech PTM, Text LLM, and Emotional TTS for Speech Emotion Recognition","date":"2023-09-19","arxiv_id":"2309.10294","repositories_listed":0,"syntology":null},{"url":null,"slug":"corpus-synthesis-for-zero-shot-asr-domain","title":"Corpus Synthesis for Zero-shot ASR domain Adaptation using Large Language Models","date":"2023-09-18","arxiv_id":"2309.10707","repositories_listed":0,"syntology":null},{"url":null,"slug":"speeding-up-speech-synthesis-in-diffusion","title":"Speech Synthesis By Unrolling Diffusion Process using Neural Network Layers","date":"2023-09-18","arxiv_id":"2309.09652","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-knowledge-distillation-via-flow","title":"Cross-lingual Knowledge Distillation via Flow-based Voice Conversion for Robust Polyglot Text-To-Speech","date":"2023-09-15","arxiv_id":"2309.08255","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxtlm-unified-decoder-only-models-for","title":"Voxtlm: unified decoder-only models for consolidating speech recognition/synthesis and speech/text continuation tasks","date":"2023-09-14","arxiv_id":"2309.07937","repositories_listed":0,"syntology":null},{"url":"/paper/cleanunet-2-a-hybrid-speech-denoising-model","slug":"cleanunet-2-a-hybrid-speech-denoising-model","title":"CleanUNet 2: A Hybrid Speech Denoising Model on Waveform and Spectrogram","date":"2023-09-12","arxiv_id":"2309.05975","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-utterance-conditioned-vae-for-speech","title":"Cross-Utterance Conditioned VAE for Speech Generation","date":"2023-09-08","arxiv_id":"2309.04156","repositories_listed":0,"syntology":null},{"url":null,"slug":"mulantts-the-microsoft-speech-synthesis","title":"MuLanTTS: The Microsoft Speech Synthesis System for Blizzard Challenge 2023","date":"2023-09-06","arxiv_id":"2309.02743","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-fruitshell-french-synthesis-system-at-the","title":"The FruitShell French synthesis system at the Blizzard 2023 Challenge","date":"2023-09-01","arxiv_id":"2309.00223","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spontaneous-style-modeling-with-semi","title":"Towards Spontaneous Style Modeling with Semi-supervised Pre-training for Conversational Text-to-Speech Synthesis","date":"2023-08-31","arxiv_id":"2308.16593","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-deepzen-speech-synthesis-system-for","title":"The DeepZen Speech Synthesis System for Blizzard Challenge 2023","date":"2023-08-30","arxiv_id":"2308.15945","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-zero-shot-speaker-adaptive","title":"Generalizable Zero-Shot Speaker Adaptive Speech Synthesis with Disentangled Representations","date":"2023-08-24","arxiv_id":"2308.13007","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokensplit-using-discrete-speech","title":"TokenSplit: Using Discrete Speech Representations for Direct, Refined, and Transcript-Conditioned Speech Separation and Recognition","date":"2023-08-21","arxiv_id":"2308.10415","repositories_listed":0,"syntology":null},{"url":null,"slug":"accurate-synthesis-of-dysarthric-speech-for","title":"Accurate synthesis of Dysarthric Speech for ASR data augmentation","date":"2023-08-16","arxiv_id":"2308.08438","repositories_listed":0,"syntology":null},{"url":null,"slug":"affectecho-speaker-independent-and-language","title":"AffectEcho: Speaker Independent and Language-Agnostic Emotion and Affect Transfer for Speech Synthesis","date":"2023-08-16","arxiv_id":"2308.08577","repositories_listed":0,"syntology":null},{"url":null,"slug":"istftnet2-faster-and-more-lightweight-istft","title":"iSTFTNet2: Faster and More Lightweight iSTFT-Based Neural Vocoder Using 1D-2D CNN","date":"2023-08-14","arxiv_id":"2308.07117","repositories_listed":0,"syntology":null},{"url":null,"slug":"expresso-a-benchmark-and-analysis-of-discrete","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","date":"2023-08-10","arxiv_id":"2308.05725","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-diffusion-models-suffer-error-propagation","title":"On Error Propagation of Diffusion Models","date":"2023-08-09","arxiv_id":"2308.05021","repositories_listed":0,"syntology":null},{"url":null,"slug":"saltts-leveraging-self-supervised-speech","title":"SALTTS: Leveraging Self-Supervised Speech Representations for improved Text-to-Speech Synthesis","date":"2023-08-02","arxiv_id":"2308.01018","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-video-to-speech-synthesis-with","title":"Audio-visual video-to-speech synthesis with synthesized input audio","date":"2023-07-31","arxiv_id":"2307.16584","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-normalizing-flows-and-diffusion","title":"Comparing normalizing flows and diffusion models for prosody and acoustic modelling in text-to-speech","date":"2023-07-31","arxiv_id":"2307.16679","repositories_listed":0,"syntology":null},{"url":null,"slug":"metts-multilingual-emotional-text-to-speech","title":"METTS: Multilingual Emotional Text-to-Speech by Cross-speaker and Cross-lingual Emotion Transfer","date":"2023-07-29","arxiv_id":"2307.15951","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimally-supervised-speech-synthesis-with","title":"Minimally-Supervised Speech Synthesis with Conditional Diffusion Model and Language Model: A Comparative Study of Semantic Coding","date":"2023-07-28","arxiv_id":"2307.15484","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-on-the-effects-of-speaker","title":"An analysis on the effects of speaker embedding choice in non auto-regressive TTS","date":"2023-07-19","arxiv_id":"2307.09898","repositories_listed":0,"syntology":null},{"url":null,"slug":"slmgan-exploiting-speech-language-model","title":"SLMGAN: Exploiting Speech Language Model Representations for Unsupervised Zero-Shot Voice Conversion in GANs","date":"2023-07-18","arxiv_id":"2307.09435","repositories_listed":0,"syntology":null},{"url":null,"slug":"mega-tts-2-zero-shot-text-to-speech-with","title":"Mega-TTS 2: Boosting Prompting Mechanisms for Zero-Shot Speech Synthesis","date":"2023-07-14","arxiv_id":"2307.07218","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-use-of-self-supervised-speech","title":"On the Use of Self-Supervised Speech Representations in Spontaneous Speech Synthesis","date":"2023-07-11","arxiv_id":"2307.05132","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustl2s-speaker-specific-lip-to-speech","title":"RobustL2S: Speaker-Specific Lip-to-Speech Synthesis exploiting Self-Supervised Representations","date":"2023-07-03","arxiv_id":"2307.01233","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-automatic-voice-over-with","title":"High-Quality Automatic Voice Over with Accurate Alignment: Supervision through Self-Supervised Discrete Speech Units","date":"2023-06-29","arxiv_id":"2306.17005","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-unsupervised-audio-pre-training","title":"Large-scale unsupervised audio pre-training for video-to-speech synthesis","date":"2023-06-27","arxiv_id":"2306.15464","repositories_listed":0,"syntology":null},{"url":null,"slug":"dse-tts-dual-speaker-embedding-for-cross","title":"DSE-TTS: Dual Speaker Embedding for Cross-Lingual Text-to-Speech","date":"2023-06-25","arxiv_id":"2306.14145","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategies-in-transfer-learning-for-low","title":"Strategies in Transfer Learning for Low-Resource Speech Synthesis: Phone Mapping, Features Input, and Source Language Selection","date":"2023-06-21","arxiv_id":"2306.12040","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-aware-text-to-speech","title":"Visual-Aware Text-to-Speech","date":"2023-06-21","arxiv_id":"2306.12020","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-prosody-transfer-for-expressive","title":"Cross-lingual Prosody Transfer for Expressive Machine Dubbing","date":"2023-06-20","arxiv_id":"2306.11658","repositories_listed":0,"syntology":null},{"url":null,"slug":"cml-tts-a-multilingual-dataset-for-speech","title":"CML-TTS A Multilingual Dataset for Speech Synthesis in Low-Resource Languages","date":"2023-06-16","arxiv_id":"2306.10097","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-utility-of-surprisal-from","title":"Investigating the Utility of Surprisal from Large Language Models for Speech Synthesis Prosody","date":"2023-06-16","arxiv_id":"2306.09814","repositories_listed":0,"syntology":null},{"url":null,"slug":"diff-ttsg-denoising-probabilistic-integrated","title":"Diff-TTSG: Denoising probabilistic integrated speech and gesture synthesis","date":"2023-06-15","arxiv_id":"2306.09417","repositories_listed":0,"syntology":null},{"url":null,"slug":"pausespeech-natural-speech-synthesis-via-pre","title":"PauseSpeech: Natural Speech Synthesis via Pre-trained Language Model and Pause-based Prosody Modeling","date":"2023-06-13","arxiv_id":"2306.07489","repositories_listed":0,"syntology":null},{"url":null,"slug":"hiddensinger-high-quality-singing-voice","title":"HiddenSinger: High-Quality Singing Voice Synthesis via Neural Audio Codec and Latent Diffusion Models","date":"2023-06-12","arxiv_id":"2306.06814","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-fast-and-high-quality-speech","title":"Boosting Fast and High-Quality Speech Synthesis with Linear Diffusion","date":"2023-06-09","arxiv_id":"2306.05708","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyvoice-language-models-for-speech-to","title":"PolyVoice: Language Models for Speech to Speech Translation","date":"2023-06-05","arxiv_id":"2306.02982","repositories_listed":0,"syntology":null},{"url":null,"slug":"rhythm-controllable-attention-with-high","title":"Rhythm-controllable Attention with High Robustness for Long Sentence Speech Synthesis","date":"2023-06-05","arxiv_id":"2306.02593","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-independent-neural-formant-synthesis","title":"Speaker-independent neural formant synthesis","date":"2023-06-02","arxiv_id":"2306.01957","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-inpainting-context-based-speech","title":"Speech inpainting: Context-based speech synthesis guided by video","date":"2023-06-01","arxiv_id":"2306.00489","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-pipeline-for-swiss-german-a","title":"Text-to-Speech Pipeline for Swiss German -- A comparison","date":"2023-05-31","arxiv_id":"2305.19750","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-evaluation-of-turn-taking-cues-in","title":"Automatic Evaluation of Turn-taking Cues in Conversational Speech Synthesis","date":"2023-05-29","arxiv_id":"2305.17971","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-personalized-synthetic-voices-from","title":"Creating Personalized Synthetic Voices from Post-Glossectomy Speech with Guided Diffusion Models","date":"2023-05-27","arxiv_id":"2305.17436","repositories_listed":0,"syntology":null},{"url":null,"slug":"lms-with-a-voice-spoken-language-modeling","title":"Spoken Question Answering and Speech Continuation Using Spectrogram-Powered LLM","date":"2023-05-24","arxiv_id":"2305.15255","repositories_listed":0,"syntology":null},{"url":null,"slug":"calls-japanese-empathetic-dialogue-speech","title":"CALLS: Japanese Empathetic Dialogue Speech Corpus of Complaint Handling and Attentive Listening in Customer Center","date":"2023-05-23","arxiv_id":"2305.13713","repositories_listed":0,"syntology":null},{"url":null,"slug":"chatgpt-edss-empathetic-dialogue-speech","title":"ChatGPT-EDSS: Empathetic Dialogue Speech Synthesis Trained from ChatGPT-derived Context Word Embeddings","date":"2023-05-23","arxiv_id":"2305.13724","repositories_listed":0,"syntology":null},{"url":null,"slug":"zet-speech-zero-shot-adaptive-emotion","title":"ZET-Speech: Zero-shot adaptive Emotion-controllable Text-to-Speech Synthesis with Diffusion and Style-based Models","date":"2023-05-23","arxiv_id":"2305.13831","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-generation-with-speech-synthesis-for-asr","title":"Text Generation with Speech Synthesis for ASR Data Augmentation","date":"2023-05-22","arxiv_id":"2305.16333","repositories_listed":0,"syntology":null},{"url":null,"slug":"vakta-setu-a-speech-to-speech-machine","title":"VAKTA-SETU: A Speech-to-Speech Machine Translation Service in Select Indic Languages","date":"2023-05-21","arxiv_id":"2305.12518","repositories_listed":0,"syntology":null},{"url":null,"slug":"mparrottts-multilingual-multi-speaker-text-to","title":"MParrotTTS: Multilingual Multi-speaker Text to Speech Synthesis in Low Resource Setting","date":"2023-05-19","arxiv_id":"2305.11926","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-front-end-framework-for-english","title":"A unified front-end framework for English text-to-speech synthesis","date":"2023-05-18","arxiv_id":"2305.10666","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-analysis-of-oral-and-nasal-vowels","title":"Empirical Analysis of Oral and Nasal Vowels of Konkani","date":"2023-05-17","arxiv_id":"2305.10122","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-personalized-lip-to-speech","title":"Zero-shot personalized lip-to-speech synthesis with face image based voice control","date":"2023-05-09","arxiv_id":"2305.14359","repositories_listed":0,"syntology":null},{"url":null,"slug":"accented-text-to-speech-synthesis-with","title":"Accented Text-to-Speech Synthesis with Limited Data","date":"2023-05-08","arxiv_id":"2305.04816","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2-ctts-end-to-end-multi-scale-multi-modal","title":"M2-CTTS: End-to-End Multi-scale Multi-modal Conversational Text-to-Speech Synthesis","date":"2023-05-03","arxiv_id":"2305.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-learning-techniques-for-3","title":"A Review of Deep Learning Techniques for Speech Processing","date":"2023-04-30","arxiv_id":"2305.00359","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-synthesis","title":"Zero-shot text-to-speech synthesis conditioned using self-supervised speech representation model","date":"2023-04-24","arxiv_id":"2304.11976","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-prosody-prediction-for-expressive","title":"Ensemble prosody prediction for expressive speech synthesis","date":"2023-04-03","arxiv_id":"2304.00714","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-is-all-you-need-personalizing-asr-models","title":"Text is All You Need: Personalizing ASR Models using Controllable Speech Synthesis","date":"2023-03-27","arxiv_id":"2303.14885","repositories_listed":0,"syntology":null},{"url":null,"slug":"wave-u-net-discriminator-fast-and-lightweight","title":"Wave-U-Net Discriminator: Fast and Lightweight Discriminator for Generative Adversarial Network-Based Speech Synthesis","date":"2023-03-24","arxiv_id":"2303.13909","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-diffusion-model-for-speech-synthesis-a","title":"A Survey on Audio Diffusion Models: Text To Speech Synthesis and Enhancement in Generative AI","date":"2023-03-23","arxiv_id":"2303.13336","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-in-speech-processing-a-survey","title":"Transformers in Speech Processing: A Survey","date":"2023-03-21","arxiv_id":"2303.11607","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-high-dimensional-data-with-sparse","title":"Controllable Prosody Generation With Partial Inputs","date":"2023-03-14","arxiv_id":"2303.09446","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-prosody-for-cross-speaker-style","title":"Improving Prosody for Cross-Speaker Style Transfer by Semi-Supervised Style Extractor and Hierarchical Modeling in Speech Synthesis","date":"2023-03-14","arxiv_id":"2303.07711","repositories_listed":0,"syntology":null},{"url":null,"slug":"qi-tts-questioning-intonation-control-for","title":"QI-TTS: Questioning Intonation Control for Emotional Speech Synthesis","date":"2023-03-14","arxiv_id":"2303.07682","repositories_listed":0,"syntology":null},{"url":null,"slug":"vani-very-lightweight-accent-controllable-tts","title":"VANI: Very-lightweight Accent-controllable TTS for Native and Non-native speakers with Identity Preservation","date":"2023-03-14","arxiv_id":"2303.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-prosody-transfer-models-transfer-prosody","title":"Do Prosody Transfer Models Transfer Prosody?","date":"2023-03-07","arxiv_id":"2303.04289","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundationtts-text-to-speech-for-asr","title":"FoundationTTS: Text-to-Speech for ASR Customization with Generative Language Model","date":"2023-03-06","arxiv_id":"2303.02939","repositories_listed":0,"syntology":null},{"url":null,"slug":"dtw-siamesenet-dynamic-time-warped-siamese","title":"DTW-SiameseNet: Dynamic Time Warped Siamese Network for Mispronunciation Detection and Correction","date":"2023-03-01","arxiv_id":"2303.00171","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-audio-visual-synchronization-for-lip","title":"On the Audio-visual Synchronization for Lip-to-Speech Synthesis","date":"2023-03-01","arxiv_id":"2303.00502","repositories_listed":0,"syntology":null},{"url":null,"slug":"parrottts-text-to-speech-synthesis-by","title":"ParrotTTS: Text-to-Speech synthesis by exploiting self-supervised representations","date":"2023-03-01","arxiv_id":"2303.01261","repositories_listed":0,"syntology":null},{"url":null,"slug":"clartts-an-open-source-classical-arabic-text","title":"ClArTTS: An Open-Source Classical Arabic Text-to-Speech Corpus","date":"2023-02-28","arxiv_id":"2303.00069","repositories_listed":0,"syntology":null},{"url":null,"slug":"crossspeech-speaker-independent-acoustic","title":"CrossSpeech: Speaker-independent Acoustic Representation for Cross-lingual Speech Synthesis","date":"2023-02-28","arxiv_id":"2302.14370","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniflg-unified-facial-landmark-generator-from","title":"UniFLG: Unified Facial Landmark Generator from Text or Speech","date":"2023-02-28","arxiv_id":"2302.14337","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-small-footprint-hybrid-hmm-hifigan","title":"Fast and small footprint Hybrid HMM-HiFiGAN based system for speech synthesis in Indian languages","date":"2023-02-13","arxiv_id":"2302.06227","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-statistical-similarity-rethinking","title":"Beyond Statistical Similarity: Rethinking Metrics for Deep Generative Models in Engineering Design","date":"2023-02-06","arxiv_id":"2302.02913","repositories_listed":0,"syntology":null},{"url":null,"slug":"uzbektagger-the-rule-based-pos-tagger-for","title":"UzbekTagger: The rule-based POS tagger for Uzbek language","date":"2023-01-30","arxiv_id":"2301.12711","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-granularity-of-prosodic-representations-in","title":"On granularity of prosodic representations in expressive text-to-speech","date":"2023-01-26","arxiv_id":"2301.11446","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-multiaccented-multispeaker-tts","title":"Multilingual Multiaccented Multispeaker TTS with RADTTS","date":"2023-01-24","arxiv_id":"2301.10335","repositories_listed":0,"syntology":null},{"url":null,"slug":"regeneration-learning-a-learning-paradigm-for","title":"Regeneration Learning: A Learning Paradigm for Data Generation","date":"2023-01-21","arxiv_id":"2301.08846","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-automated-machine-translation-to","title":"Applying Automated Machine Translation to Educational Video Courses","date":"2023-01-09","arxiv_id":"2301.03141","repositories_listed":0,"syntology":null},{"url":null,"slug":"revise-self-supervised-speech-resynthesis-1","title":"ReVISE: Self-Supervised Speech Resynthesis With Visual Input for Universal and Generalized Speech Regeneration","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hmm-based-data-augmentation-for-e2e-systems","title":"HMM-based data augmentation for E2E systems for building conversational speech synthesis systems","date":"2022-12-22","arxiv_id":"2212.11982","repositories_listed":0,"syntology":null},{"url":"/paper/revise-self-supervised-speech-resynthesis","slug":"revise-self-supervised-speech-resynthesis","title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","date":"2022-12-21","arxiv_id":"2212.11377","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-japanese-png-bert-language","title":"Investigation of Japanese PnG BERT language model in text-to-speech synthesis for pitch accent language","date":"2022-12-16","arxiv_id":"2212.08321","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-based-on-latent","title":"Text-to-speech synthesis based on latent variable conversion using diffusion probabilistic model and variational autoencoder","date":"2022-12-16","arxiv_id":"2212.08329","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-label-free-cross-speaker-style-transfer","title":"Style-Label-Free: Cross-Speaker Style Transfer by Quantized VAE and Speaker-wise Normalization in Speech Synthesis","date":"2022-12-13","arxiv_id":"2212.06397","repositories_listed":0,"syntology":null},{"url":null,"slug":"snac-speaker-normalized-affine-coupling-layer","title":"SNAC: Speaker-normalized affine coupling layer in flow-based architecture for zero-shot multi-speaker text-to-speech","date":"2022-11-30","arxiv_id":"2211.16866","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-speech-synthesis-by-learning","title":"Controllable speech synthesis by learning discrete phoneme-level prosodic representations","date":"2022-11-29","arxiv_id":"2211.16307","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-expressive-text-to-speech","title":"Contextual Expressive Text-to-Speech","date":"2022-11-26","arxiv_id":"2211.14548","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-incremental-text-to-speech-on-gpus","title":"Efficient Incremental Text-to-Speech on GPUs","date":"2022-11-25","arxiv_id":"2211.13939","repositories_listed":0,"syntology":null},{"url":null,"slug":"la-voce-low-snr-audio-visual-speech","title":"LA-VocE: Low-SNR Audio-visual Speech Enhancement using Neural Vocoders","date":"2022-11-20","arxiv_id":"2211.10999","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-expressive-speech-synthesis-via","title":"Multi-Speaker Expressive Speech Synthesis via Multiple Factors Decoupling","date":"2022-11-19","arxiv_id":"2211.10568","repositories_listed":0,"syntology":null}],"record_sha256":"c88a06e3fe55b4d19a450d9334c58635bafc078dcd149622d1fa1ba3b2e7f6dc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}