{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech/papers/11","list_of":"/task/text-to-speech","task":"Text to Speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":15,"rows_per_page":100,"rows":[1001,1100],"of":1419,"counts":{"archive_papers_tagged":1419,"with_a_code_link":399,"where_syntology_ran_a_sample":108,"not_listed_spam_title":0,"listed":1419,"listed_where_code_ran":108,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech","prev":"/task/text-to-speech/papers/10","next":"/task/text-to-speech/papers/12","papers":[{"url":null,"slug":"building-open-source-speech-technology-for","title":"Building Open-source Speech Technology for Low-resource Minority Languages with SáMi as an Example – Tools, Methods and Experiments","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"error-annotation-in-post-editing-machine","title":"Error Annotation in Post-Editing Machine Translation: Investigating the Impact of Text-to-Speech Technology","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-transfer-learning-for-urdu-speech","title":"Exploring Transfer Learning for Urdu Speech Synthesis","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"huqariq-a-multilingual-speech-corpus-of-1","title":"Huqariq: A Multilingual Speech Corpus of Native Languages of Peru forSpeech Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-inter-and-intra-speaker-voice","title":"Investigating Inter- and Intra-speaker Voice Conversion using Audiobooks","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"parlamentparla-a-speech-corpus-of-catalan","title":"ParlamentParla: A Speech Corpus of Catalan Parliamentary Sessions","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reading-assistance-through-lara-the-learning","title":"Reading Assistance through LARA, the Learning And Reading Assistant","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-for-under-resourced-languages","title":"Text-to-Speech for Under-Resourced Languages: Phoneme Mapping and Source Language Selection in Transfer Learning","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nos-project-opening-routes-for-the","title":"The Nós Project: Opening routes for the Galician language in the field of language technologies","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-the-lara-little-prince-to-compare-human","title":"Using the LARA Little Prince to compare human and TTS audio quality","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-2-a-diffusion-model-for-high","title":"Guided-TTS 2: A Diffusion Model for High-quality Adaptive Text-to-Speech with Untranscribed Data","date":"2022-05-30","arxiv_id":"2205.15370","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-transliterated-words-for-finding","title":"Exploiting Transliterated Words for Finding Similarity in Inter-Language News Articles using Machine Learning","date":"2022-05-29","arxiv_id":"2206.11860","repositories_listed":0,"syntology":null},{"url":null,"slug":"t-modules-translation-modules-for-zero-shot","title":"T-Modules: Translation Modules for Zero-Shot Cross-Modal Machine Translation","date":"2022-05-24","arxiv_id":"2205.12216","repositories_listed":0,"syntology":null},{"url":"/paper/talking-face-generation-with-multilingual-tts","slug":"talking-face-generation-with-multilingual-tts","title":"Talking Face Generation with Multilingual TTS","date":"2022-05-13","arxiv_id":"2205.06421","repositories_listed":0,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/talking-face-generation-with-multilingual-tts#ran","syntology_url":"https://syntology.ai/paper/2205.06421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.06421"}},"official":null}},{"url":null,"slug":"recab-vae-gumbel-softmax-variational","title":"ReCAB-VAE: Gumbel-Softmax Variational Inference Based on Analytic Divergence","date":"2022-05-09","arxiv_id":"2205.04104","repositories_listed":0,"syntology":null},{"url":null,"slug":"regotron-regularizing-the-tacotron2","title":"Regotron: Regularizing the Tacotron2 architecture via monotonic alignment loss","date":"2022-04-28","arxiv_id":"2204.13437","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-speaker-emotion-transfer-for-low","title":"Cross-Speaker Emotion Transfer for Low-Resource Text-to-Speech Using Non-Parallel Voice Conversion with Pitch-Shift Data Augmentation","date":"2022-04-21","arxiv_id":"2204.10020","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-deep-fake-detection-system-with-neural","title":"Audio Deep Fake Detection System with Neural Stitching for ADD 2022","date":"2022-04-19","arxiv_id":"2204.08720","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-feature-underspecified-lexicon","title":"Applying Feature Underspecified Lexicon Phonological Features in Multilingual Text-to-Speech","date":"2022-04-14","arxiv_id":"2204.07228","repositories_listed":0,"syntology":null},{"url":null,"slug":"study-of-indian-english-pronunciation","title":"Study of Indian English Pronunciation Variabilities relative to Received Pronunciation","date":"2022-04-13","arxiv_id":"2204.06502","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancement-of-pitch-controllability-using","title":"Enhancement of Pitch Controllability using Timbre-Preserving Pitch Augmentation in FastPitch","date":"2022-04-12","arxiv_id":"2204.05753","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-noise-control-for-multispeaker","title":"Fine-grained Noise Control for Multispeaker Speech Synthesis","date":"2022-04-11","arxiv_id":"2204.05070","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-partialspoof-database-and-countermeasures","title":"The PartialSpoof Database and Countermeasures for the Detection of Short Fake Speech Segments Embedded in an Utterance","date":"2022-04-11","arxiv_id":"2204.05177","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-and-multi-scale-variational","title":"Hierarchical and Multi-Scale Variational Autoencoder for Diverse and Natural Non-Autoregressive Text-to-Speech","date":"2022-04-08","arxiv_id":"2204.04004","repositories_listed":0,"syntology":null},{"url":null,"slug":"karaoker-alignment-free-singing-voice","title":"Karaoker: Alignment-free singing voice synthesis with speech training data","date":"2022-04-08","arxiv_id":"2204.04127","repositories_listed":0,"syntology":null},{"url":null,"slug":"arabic-text-to-speech-tts-data-preparation","title":"Arabic Text-To-Speech (TTS) Data Preparation","date":"2022-04-07","arxiv_id":"2204.03255","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-quantized-prosody-representation","title":"Unsupervised Quantized Prosody Representation for Controllable Speech Synthesis","date":"2022-04-07","arxiv_id":"2204.03238","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-direct-speech-to-speech-translation","title":"Enhanced Direct Speech-to-Speech Translation Using Self-supervised Pre-training and Data Augmentation","date":"2022-04-06","arxiv_id":"2204.02967","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-selective-self-distillation","title":"Representation Selective Self-distillation and wav2vec 2.0 Feature Exploration for Spoof-aware Speaker Verification","date":"2022-04-06","arxiv_id":"2204.02639","repositories_listed":0,"syntology":null},{"url":"/paper/somos-the-samsung-open-mos-dataset-for-the","slug":"somos-the-samsung-open-mos-dataset-for-the","title":"SOMOS: The Samsung Open MOS Dataset for the Evaluation of Neural Text-to-Speech Synthesis","date":"2022-04-06","arxiv_id":"2204.03040","repositories_listed":0,"syntology":null},{"url":null,"slug":"anti-spoofing-using-transfer-learning-with","title":"Anti-Spoofing Using Transfer Learning with Variational Information Bottleneck","date":"2022-04-04","arxiv_id":"2204.01387","repositories_listed":0,"syntology":null},{"url":null,"slug":"deliberation-model-for-on-device-spoken","title":"Deliberation Model for On-Device Spoken Language Understanding","date":"2022-04-04","arxiv_id":"2204.01893","repositories_listed":0,"syntology":null},{"url":null,"slug":"vqtts-high-fidelity-text-to-speech-synthesis","title":"VQTTS: High-Fidelity Text-to-Speech Synthesis with Self-Supervised VQ Acoustic Feature","date":"2022-04-02","arxiv_id":"2204.00768","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaspeech-4-adaptive-text-to-speech-in-zero","title":"AdaSpeech 4: Adaptive Text to Speech in Zero-Shot Scenarios","date":"2022-04-01","arxiv_id":"2204.00436","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-data-augmentation-for-low","title":"Text-To-Speech Data Augmentation for Low Resource Speech Recognition","date":"2022-04-01","arxiv_id":"2204.00291","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectiveness-of-text-to-speech-pseudo-labels","title":"Effectiveness of text to speech pseudo labels for forced alignment and cross lingual pretrained models for low resource speech recognition","date":"2022-03-31","arxiv_id":"2203.16823","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-phoneme-bert-improving-bert-with-mixed","title":"Mixed-Phoneme BERT: Improving BERT with Mixed Phoneme and Sup-Phoneme Representations for Text to Speech","date":"2022-03-31","arxiv_id":"2203.17190","repositories_listed":0,"syntology":null},{"url":"/paper/open-source-magicdata-ramc-a-rich-annotated","slug":"open-source-magicdata-ramc-a-rich-annotated","title":"Open Source MagicData-RAMC: A Rich Annotated Mandarin Conversational(RAMC) Speech Dataset","date":"2022-03-31","arxiv_id":"2203.16844","repositories_listed":0,"syntology":null},{"url":null,"slug":"wavthruvec-latent-speech-representation-as","title":"WavThruVec: Latent speech representation as intermediate features for neural speech synthesis","date":"2022-03-31","arxiv_id":"2203.16930","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-audio-deepfake-detection-generalize","title":"Does Audio Deepfake Detection Generalize?","date":"2022-03-30","arxiv_id":"2203.16263","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-syntax-unicode-x2013-prosody-mapping","title":"Applying Syntax$\\unicode{x2013}$Prosody Mapping Hypothesis and Prosodic Well-Formedness Constraints to Neural Sequence-to-Sequence Speech Synthesis","date":"2022-03-29","arxiv_id":"2203.15276","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-framework-for-low-resource","title":"Transfer Learning Framework for Low-Resource Text-to-Speech using a Large-Scale Unlabeled Speech Corpus","date":"2022-03-29","arxiv_id":"2203.15447","repositories_listed":0,"syntology":null},{"url":null,"slug":"studies-corpus-of-japanese-empathetic","title":"STUDIES: Corpus of Japanese Empathetic Dialogue Speech Towards Friendly Voice Agent","date":"2022-03-28","arxiv_id":"2203.14757","repositories_listed":0,"syntology":null},{"url":null,"slug":"bunched-lpcnet2-efficient-neural-vocoders","title":"Bunched LPCNet2: Efficient Neural Vocoders Covering Devices from Cloud to Edge","date":"2022-03-27","arxiv_id":"2203.14416","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-text-to-speech-pipeline-evaluation","title":"A Text-to-Speech Pipeline, Evaluation Methodology, and Initial Fine-Tuning Results for Child Speech Synthesis","date":"2022-03-22","arxiv_id":"2203.11562","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-duration-modeling-for-end-to","title":"AutoTTS: End-to-End Text-to-Speech Synthesis through Differentiable Duration Modeling","date":"2022-03-21","arxiv_id":"2203.11049","repositories_listed":0,"syntology":null},{"url":null,"slug":"vocal-effort-modeling-in-neural-tts-for","title":"Vocal effort modeling in neural TTS for improving the intelligibility of synthetic speech in noise","date":"2022-03-20","arxiv_id":"2203.10637","repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-few-shot-voice-cloning-using-multi","title":"Improve few-shot voice cloning using multi-modal learning","date":"2022-03-18","arxiv_id":"2203.09708","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-free-non-parallel-many-to-many-voice","title":"Text-free non-parallel many-to-many voice conversion using normalising flows","date":"2022-03-15","arxiv_id":"2203.08009","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-over-smoothness-in-text-to-speech","title":"Revisiting Over-Smoothness in Text to Speech","date":"2022-02-26","arxiv_id":"2202.13066","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-detection-of-political-deepfakes-across","title":"Human Detection of Political Speech Deepfakes across Transcripts, Audio, and Video","date":"2022-02-25","arxiv_id":"2202.12883","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-lingual-speech-synthesis-with","title":"Improving Cross-lingual Speech Synthesis with Triplet Training Scheme","date":"2022-02-22","arxiv_id":"2202.10729","repositories_listed":0,"syntology":null},{"url":null,"slug":"r-g2p-evaluating-and-enhancing-robustness-of","title":"r-G2P: Evaluating and Enhancing Robustness of Grapheme to Phoneme Conversion by Controlled noise introducing and Contextual information incorporation","date":"2022-02-21","arxiv_id":"2202.11194","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosospeech-enhancing-prosody-with-quantized","title":"ProsoSpeech: Enhancing Prosody With Quantized Vector Pre-training in Text-to-Speech","date":"2022-02-16","arxiv_id":"2202.07816","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-filter-few-shot-text-to-speech-speaker","title":"Voice Filter: Few-shot text-to-speech speaker adaptation using voice conversion as a post-processing module","date":"2022-02-16","arxiv_id":"2202.08164","repositories_listed":0,"syntology":null},{"url":null,"slug":"newspod-automatic-and-interactive-news","title":"NewsPod: Automatic and Interactive News Podcasts","date":"2022-02-15","arxiv_id":"2202.07146","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-word-level-prosody-tagging-for","title":"Unsupervised word-level prosody tagging for controllable speech synthesis","date":"2022-02-15","arxiv_id":"2202.07200","repositories_listed":0,"syntology":null},{"url":null,"slug":"distribution-augmentation-for-low-resource","title":"Distribution augmentation for low-resource expressive text-to-speech","date":"2022-02-13","arxiv_id":"2202.06409","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-performer-score-to-audio-music","title":"Deep Performer: Score-to-Audio Music Performance Synthesis","date":"2022-02-12","arxiv_id":"2202.06034","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-speaker-style-transfer-for-text-to","title":"Cross-speaker style transfer for text-to-speech using data augmentation","date":"2022-02-10","arxiv_id":"2202.05083","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-synthetic-speaker-profiles-in-text","title":"Building Synthetic Speaker Profiles in Text-to-Speech Systems","date":"2022-02-07","arxiv_id":"2202.03125","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-deep-transfer-learning-for-emiot","title":"Multi-Stage Deep Transfer Learning for EmIoT-enabled Human-Computer Interaction","date":"2022-02-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-models-of-text","title":"Transformer-based Models of Text Normalization for Speech Applications","date":"2022-02-01","arxiv_id":"2202.00153","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-dysarthric-speech-using-multi","title":"Synthesizing Dysarthric Speech Using Multi-talker TTS for Dysarthric Speech Recognition","date":"2022-01-27","arxiv_id":"2201.11571","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-msxf-tts-system-for-icassp-2022-add","title":"The MSXF TTS System for ICASSP 2022 ADD Challenge","date":"2022-01-27","arxiv_id":"2201.11400","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-long-form-voice-cloning-with","title":"Zero-Shot Long-Form Voice Cloning with Dynamic Convolution Attention","date":"2022-01-25","arxiv_id":"2201.10375","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyphone-disambiguation-and-accent","title":"Polyphone disambiguation and accent prediction using pre-trained language models in Japanese TTS front-end","date":"2022-01-24","arxiv_id":"2201.09427","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-text-to-speech-using-multi-task","title":"Cross-Lingual Text-to-Speech Using Multi-Task Learning and Speaker Classifier Joint Training","date":"2022-01-20","arxiv_id":"2201.08124","repositories_listed":0,"syntology":null},{"url":null,"slug":"empathic-machines-using-intermediate-features","title":"Empathic Machines: Using Intermediate Features as Levers to Emulate Emotions in Text-To-Speech Systems","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kazakhtts2-extending-the-open-source-kazakh","title":"KazakhTTS2: Extending the Open-Source Kazakh TTS Corpus With More Data, Speakers, and Topics","date":"2022-01-15","arxiv_id":"2201.05771","repositories_listed":0,"syntology":null},{"url":null,"slug":"sok-a-study-of-the-security-on-voice","title":"SoK: A Study of the Security on Voice Processing Systems","date":"2021-12-24","arxiv_id":"2112.13144","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-multi-style-text-to-speech","title":"Multi-speaker Multi-style Text-to-speech Synthesis With Single-speaker Single-style Training Data Scenarios","date":"2021-12-23","arxiv_id":"2112.12743","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-emotional-text-to-speech","title":"Multi-speaker Emotional Text-to-speech Synthesizer","date":"2021-12-07","arxiv_id":"2112.03557","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-t-transducer-for-text-to-speech-and","title":"Speech-T: Transducer for Text to Speech and Beyond","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-rich-product-descriptions-for","title":"Generating Rich Product Descriptions for Conversational E-commerce Systems","date":"2021-11-30","arxiv_id":"2111.15298","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-text-to-speech-with-untranscribed-1","title":"Guided-TTS: A Diffusion Model for Text-to-Speech via Classifier Guidance","date":"2021-11-23","arxiv_id":"2111.11755","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-prosodic-clustering-for-multispeaker","title":"Improved Prosodic Clustering for Multispeaker and Speaker-independent Phoneme-level Prosody Control","date":"2021-11-19","arxiv_id":"2111.10168","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodic-clustering-for-phoneme-level-prosody","title":"Prosodic Clustering for Phoneme-level Prosody Control in End-to-End Speech Synthesis","date":"2021-11-19","arxiv_id":"2111.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-transfer-learning-for","title":"Semi-supervised transfer learning for language expansion of end-to-end speech recognition models to low-resource languages","date":"2021-11-19","arxiv_id":"2111.10047","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-streaming-speech-synthesis-with","title":"High Quality Streaming Speech Synthesis with Low, Sentence-Length-Independent Latency","date":"2021-11-17","arxiv_id":"2111.09052","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-utterance-conditioned-vae-for-non","title":"Cross-Utterance Conditioned VAE for Non-Autoregressive Text-to-Speech","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-for-low-resource-languages","title":"Speech Synthesis for Low Resource Languages using Transliteration Enabled Transfer Learning","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-voice-fast-few-shot-style-transfer-for","title":"Meta-Voice: Fast few-shot style transfer for expressive voice cloning using meta learning","date":"2021-11-14","arxiv_id":"2111.07218","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotional-prosody-control-for-speech","title":"Emotional Prosody Control for Speech Generation","date":"2021-11-07","arxiv_id":"2111.04730","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-generation","title":"Speaker Generation","date":"2021-11-07","arxiv_id":"2111.05095","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-prosody-in-end-to-end-tts-a-case","title":"Controlling Prosody in End-to-End TTS: A Case Study on Contrastive Focus Generation","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vida-man-visual-dialog-with-digital-humans","title":"ViDA-MAN: Visual Dialog with Digital Humans","date":"2021-10-26","arxiv_id":"2110.13384","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-acoustic-space-for-an-efficient","title":"Discrete Acoustic Space for an Efficient Sampling in Neural Text-To-Speech","date":"2021-10-24","arxiv_id":"2110.12539","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-speech-synthesis-for-speech-to","title":"From Start to Finish: Latency Reduction Strategies for Incremental Speech Synthesis in Simultaneous Speech-to-Speech Translation","date":"2021-10-15","arxiv_id":"2110.08214","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-dubber-dubbing-for-silent-videos","title":"Neural Dubber: Dubbing for Videos According to Scripts","date":"2021-10-15","arxiv_id":"2110.08243","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-timbre-disentanglement-in-non","title":"Exploring Timbre Disentanglement in Non-Autoregressive Cross-Lingual Text-to-Speech","date":"2021-10-14","arxiv_id":"2110.07192","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedspeech-federated-text-to-speech-with","title":"FedSpeech: Federated Text-to-Speech with Continual Learning","date":"2021-10-14","arxiv_id":"2110.07216","repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-cross-lingual-voice-cloning-using-low","title":"Improve Cross-lingual Voice Cloning Using Low-quality Code-switched Data","date":"2021-10-14","arxiv_id":"2110.07210","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-ipa-based-cross-lingual-text-to","title":"Revisiting IPA-based Cross-lingual Text-to-speech","date":"2021-10-14","arxiv_id":"2110.07187","repositories_listed":0,"syntology":null},{"url":null,"slug":"singgan-generative-adversarial-network-for","title":"SingGAN: Generative Adversarial Network For High-Fidelity Singing Voice Generation","date":"2021-10-14","arxiv_id":"2110.07468","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-melody-unsupervision-model-for-singing","title":"A Melody-Unsupervision Model for Singing Voice Synthesis","date":"2021-10-13","arxiv_id":"2110.06546","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-tts-models-for-new-speakers-using","title":"Adapting TTS models For New Speakers using Transfer Learning","date":"2021-10-12","arxiv_id":"2110.05798","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-efficacy-of-model-pre-training","title":"A study on the efficacy of model pre-training in developing neural text-to-speech system","date":"2021-10-08","arxiv_id":"2110.03857","repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-aware-text-to-speech-synthesis","title":"Environment Aware Text-to-Speech Synthesis","date":"2021-10-08","arxiv_id":"2110.03887","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualtts-tts-with-accurate-lip-speech","title":"VisualTTS: TTS with Accurate Lip-Speech Synchronization for Automatic Voice Over","date":"2021-10-07","arxiv_id":"2110.03342","repositories_listed":0,"syntology":null}],"record_sha256":"4f819140b774cb7609ab6af4014ccba39577b6cf29c50ac3aa37adf07147b6c3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}