{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/10","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":13,"rows_per_page":100,"rows":[901,1000],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/9","next":"/task/speech-synthesis/papers/11","papers":[{"url":null,"slug":"training-wake-word-detection-with-synthesized","title":"Training Wake Word Detection with Synthesized Speech Data on Confusion Words","date":"2020-11-03","arxiv_id":"2011.01460","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptually-guided-end-to-end-text-to-speech","title":"Learning to Maximize Speech Quality Directly Using MOS Prediction for Neural Text-to-Speech","date":"2020-11-02","arxiv_id":"2011.01174","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-translation","title":"Simultaneous Translation","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"emotion-controllable-speech-synthesis-using","title":"Emotion controllable speech synthesis using emotion-unlabeled dataset with the assistance of cross-domain speech emotion recognition","date":"2020-10-26","arxiv_id":"2010.13350","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stream-attention-based-blstm-with","title":"Multi-stream Attention-based BLSTM with Feature Segmentation for Speech Emotion Recognition","date":"2020-10-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"graphspeech-syntax-aware-graph-attention","title":"GraphSpeech: Syntax-Aware Graph Attention Network For Neural Speech Synthesis","date":"2020-10-23","arxiv_id":"2010.12423","repositories_listed":0,"syntology":null},{"url":null,"slug":"nu-gan-high-resolution-neural-upsampling-with","title":"NU-GAN: High resolution neural upsampling with GAN","date":"2020-10-22","arxiv_id":"2010.11362","repositories_listed":0,"syntology":null},{"url":null,"slug":"grapheme-or-phoneme-an-analysis-of-tacotron-s","title":"An Investigation of the Relation Between Grapheme Embeddings and Pronunciation for Tacotron-based Systems","date":"2020-10-21","arxiv_id":"2010.10694","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-text-to-speech-using-latent","title":"End-to-End Text-to-Speech using Latent Duration based on VQ-VAE","date":"2020-10-19","arxiv_id":"2010.09602","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-zero-resource-speech-challenge-2020","title":"The Zero Resource Speech Challenge 2020: Discovering discrete subword and word units","date":"2020-10-12","arxiv_id":"2010.05967","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-speech-synthesis-for-estonian","title":"Neural Speech Synthesis for Estonian","date":"2020-10-06","arxiv_id":"2010.02636","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-arabic-dialect-identification","title":"Automatic Arabic Dialect Identification Systems for Written Texts: A Survey","date":"2020-09-26","arxiv_id":"2009.12622","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-face-models-for-example-based-visual","title":"Neural Face Models for Example-Based Visual Speech Synthesis","date":"2020-09-22","arxiv_id":"2009.10361","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-artifacts-in-eeg-signals-for","title":"Analysis of artifacts in EEG signals for building BCIs","date":"2020-09-18","arxiv_id":"2009.09116","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-multi-grained-generative-model","title":"Hierarchical Multi-Grained Generative Model for Expressive Speech Synthesis","date":"2020-09-17","arxiv_id":"2009.08474","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-spoken-digit-generation-with","title":"Conditional Spoken Digit Generation with StyleGAN","date":"2020-09-15","arxiv_id":"2004.13764","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-neural-text-to-speech-synthesis","title":"Controllable neural text-to-speech synthesis using intuitive prosodic features","date":"2020-09-14","arxiv_id":"2009.06775","repositories_listed":0,"syntology":null},{"url":null,"slug":"corrective-feedback-emphatic-speech-synthesis","title":"Visual-speech Synthesis of Exaggerated Corrective Feedback","date":"2020-09-12","arxiv_id":"2009.05748","repositories_listed":0,"syntology":null},{"url":null,"slug":"silent-speech-interfaces-for-speech","title":"Silent Speech Interfaces for Speech Restoration: A Review","date":"2020-09-04","arxiv_id":"2009.02110","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-the-future-brings-investigating-the","title":"What the Future Brings: Investigating the Impact of Lookahead for Incremental Neural TTS","date":"2020-09-04","arxiv_id":"2009.02035","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-by-cascading-automatic","title":"Voice Conversion by Cascading Automatic Speech Recognition and Text-to-Speech Synthesis with Prosody Transfer","date":"2020-09-03","arxiv_id":"2009.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-study-on-deep-learning-based","title":"A Preliminary Study on Deep Learning-based Chinese Text to Taiwanese Speech Synthesis System","date":"2020-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-single-speaker-taiwanese-accented","title":"Real-Time Single-Speaker Taiwanese-Accented Mandarin Speech Synthesis System","date":"2020-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-neural-speech-synthesis-for-low","title":"Efficient neural speech synthesis for low-resource languages through multilingual modeling","date":"2020-08-20","arxiv_id":"2008.09659","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-learning-mechanism-for-speech","title":"Prosody Learning Mechanism for Speech Synthesis System Without Text Length Limit","date":"2020-08-13","arxiv_id":"2008.05656","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-prosodic-phrasing-with-multi-task","title":"Modeling Prosodic Phrasing with Multi-Task Learning in Tacotron-based TTS","date":"2020-08-11","arxiv_id":"2008.05284","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-mos-predictor-for-synthetic-speech-using","title":"Deep MOS Predictor for Synthetic Speech Using Cluster-Based Modeling","date":"2020-08-09","arxiv_id":"2008.03710","repositories_listed":0,"syntology":null},{"url":null,"slug":"lrspeech-extremely-low-resource-speech","title":"LRSpeech: Extremely Low-Resource Speech Synthesis and Recognition","date":"2020-08-09","arxiv_id":"2008.03687","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-machine-of-few-words-interactive-speaker","title":"A Machine of Few Words -- Interactive Speaker Recognition with Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03127","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-neural-prosody-synthesis","title":"Controllable Neural Prosody Synthesis","date":"2020-08-07","arxiv_id":"2008.03388","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-text-to-speech-synthesis-using","title":"Multi-speaker Text-to-speech Synthesis Using Deep Gaussian Processes","date":"2020-08-07","arxiv_id":"2008.02950","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-boost-by-exploiting-the-auxiliary","title":"Learning Boost by Exploiting the Auxiliary Task in Multi-task Domain","date":"2020-08-05","arxiv_id":"2008.02043","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiovisual-speech-synthesis-using-tacotron2","title":"Audiovisual Speech Synthesis using Tacotron2","date":"2020-08-03","arxiv_id":"2008.00620","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-transfer-learning-end-to-end-arabictext-to","title":"A Transfer Learning End-to-End ArabicText-To-Speech (TTS) Deep Architecture","date":"2020-07-22","arxiv_id":"2007.11541","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodic-prominence-and-boundaries-in","title":"Prosodic Prominence and Boundaries in Sequence-to-Sequence Speech Synthesis","date":"2020-06-29","arxiv_id":"2006.15967","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalizing-text-using-language-modelling","title":"Normalizing Text using Language Modelling based on Phonetics and String Similarity","date":"2020-06-25","arxiv_id":"2006.14116","repositories_listed":0,"syntology":null},{"url":null,"slug":"articulatory-wavenet-autoregressive-model-for","title":"Articulatory-WaveNet: Autoregressive Model For Acoustic-to-Articulatory Inversion","date":"2020-06-22","arxiv_id":"2006.12594","repositories_listed":0,"syntology":null},{"url":"/paper/se-melgan-speaker-agnostic-rapid-speech","slug":"se-melgan-speaker-agnostic-rapid-speech","title":"SE-MelGAN -- Speaker Agnostic Rapid Speech Enhancement","date":"2020-06-13","arxiv_id":"2006.07637","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-and-synthesis-of-hypo-and","title":"Analysis and Synthesis of Hypo and Hyperarticulated Speech","date":"2020-06-07","arxiv_id":"2006.04136","repositories_listed":0,"syntology":null},{"url":null,"slug":"etude-comparative-des-param-etres-d-entr-ee","title":"\\'Etude comparative des param\\`etres d'entr\\'ee pour la synth\\`ese expressive audiovisuelle de la parole par DNNs (Comparative study of input parameters for DNN-based expressive audiovisual speech synthesis )","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-vietnamese-statistical","title":"A comparison of Vietnamese Statistical Parametric Speech Synthesis Systems","date":"2020-05-26","arxiv_id":"2005.12962","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-tts-for-low-resource-speakers","title":"Noise Robust TTS for Low Resource Speakers using Pre-trained Model and Speech Enhancement","date":"2020-05-26","arxiv_id":"2005.12531","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-vq-vae-for-unsupervised-unit","title":"Transformer VQ-VAE for Unsupervised Unit Discovery and Speech Synthesis: ZeroSpeech 2020 Challenge","date":"2020-05-24","arxiv_id":"2005.11676","repositories_listed":0,"syntology":null},{"url":null,"slug":"nautilus-a-versatile-voice-cloning-system","title":"NAUTILUS: a Versatile Voice Cloning System","date":"2020-05-22","arxiv_id":"2005.11004","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-multispeaker-text-to-speech","title":"Cross-lingual Multispeaker Text-to-Speech under Limited-Data Scenario","date":"2020-05-21","arxiv_id":"2005.10441","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-features-and-metrics-for-high","title":"Evaluating Features and Metrics for High-Quality Simulation of Early Vocal Learning of Vowels","date":"2020-05-20","arxiv_id":"2005.09986","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-learning-abilities-on","title":"Investigation of learning abilities on linguistic features in sequence-to-sequence text-to-speech synthesis","date":"2020-05-20","arxiv_id":"2005.10390","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-subspace-hmm-for-the-zerospeech-2020","title":"Bayesian Subspace HMM for the Zerospeech 2020 Challenge","date":"2020-05-19","arxiv_id":"2005.09282","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-for-multi-speaker","title":"Semi-supervised Learning for Multi-speaker Text-to-speech Synthesis Using Discrete Speech Representation","date":"2020-05-16","arxiv_id":"2005.08024","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-do-not-need-more-data-improving-end-to","title":"You Do Not Need More Data: Improving End-To-End Speech Recognition by Text-To-Speech Data Augmentation","date":"2020-05-14","arxiv_id":"2005.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-user-centric-and-content-driven","title":"Building A User-Centric and Content-Driven Socialbot","date":"2020-05-06","arxiv_id":"2005.02623","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-prompt-selection-for-evaluation-of","title":"Augmented Prompt Selection for Evaluation of Spontaneous Speech Synthesis","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"corpus-generation-for-voice-command-in-smart","title":"Corpus Generation for Voice Command in Smart Home and the Effect of Speech Synthesis on End-to-End SLU","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"development-and-evaluation-of-speech","title":"Development and Evaluation of Speech Synthesis Corpora for Latvian","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dnn-based-speech-synthesis-using-abundant","title":"DNN-based Speech Synthesis Using Abundant Tags of Spontaneous Speech Corpus","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"manual-speech-synthesis-data-acquisition-from","title":"Manual Speech Synthesis Data Acquisition - From Script Design to Recording Speech","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-text-to-speech-synthesis-for-an-under","title":"Neural Text-to-Speech Synthesis for an Under-Resourced Language in a Diglossic Environment: the Case of Gascon Occitan","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"open-source-multi-speaker-speech-corpora-for","title":"Open-source Multi-speaker Speech Corpora for Building Gujarati, Kannada, Malayalam, Marathi, Tamil and Telugu Speech Synthesis Systems","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"style-variation-as-a-vantage-point-for-code","title":"Style Variation as a Vantage Point for Code-Switching","date":"2020-05-01","arxiv_id":"2005.00458","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-feature-learning-and-unsupervised","title":"Adversarial Feature Learning and Unsupervised Clustering based Speech Synthesis for Found Data with Acoustic and Textual Noise","date":"2020-04-28","arxiv_id":"2004.13595","repositories_listed":0,"syntology":null},{"url":null,"slug":"utterance-level-sequential-modeling-for-deep","title":"Utterance-level Sequential Modeling For Deep Gaussian Process Based Speech Synthesis Using Simple Recurrent Unit","date":"2020-04-22","arxiv_id":"2004.10823","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-explainability-study-of-the-constant-q","title":"An explainability study of the constant Q cepstral coefficient spoofing countermeasure for automatic speaker verification","date":"2020-04-19","arxiv_id":"2004.06422","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-speech-synthesis-using-eeg","title":"Advancing Speech Synthesis using EEG","date":"2020-04-09","arxiv_id":"2004.04731","repositories_listed":0,"syntology":null},{"url":null,"slug":"vocoder-based-speech-synthesis-from-silent","title":"Vocoder-Based Speech Synthesis from Silent Videos","date":"2020-04-06","arxiv_id":"2004.02541","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-style-and-content-separation-by","title":"Unsupervised Style and Content Separation by Minimizing Mutual Information for Speech Synthesis","date":"2020-03-09","arxiv_id":"2003.06227","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphtts-graph-to-sequence-modelling-in","title":"GraphTTS: graph-to-sequence modelling in neural text-to-speech","date":"2020-03-04","arxiv_id":"2003.01924","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-eeg-features-from-acoustic","title":"Generating EEG features from Acoustic features","date":"2020-02-29","arxiv_id":"2003.00007","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-sequence-to-sequence-neural-tts","title":"Controllable Sequence-To-Sequence Neural TTS with LPCNET Backend for Real-time Speech Synthesis on CPU","date":"2020-02-25","arxiv_id":"2002.10708","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-using-eeg","title":"Speech Synthesis using EEG","date":"2020-02-22","arxiv_id":"2002.12756","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-hierarchical-fine-grained-prosody","title":"Fully-hierarchical fine-grained prosody modeling for interpretable speech synthesis","date":"2020-02-06","arxiv_id":"2002.03785","repositories_listed":0,"syntology":null},{"url":null,"slug":"eigenresiduals-for-improved-parametric-speech","title":"Eigenresiduals for improved Parametric Speech Synthesis","date":"2020-01-02","arxiv_id":"2001.00581","repositories_listed":0,"syntology":null},{"url":null,"slug":"excitation-based-voice-quality-analysis-and","title":"Excitation-based Voice Quality Analysis and Modification","date":"2020-01-02","arxiv_id":"2001.00582","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-a-pitch-synchronous-residual-codebook","title":"Using a Pitch-Synchronous Residual Codebook for Hybrid HMM/Frame Selection Speech Synthesis","date":"2019-12-30","arxiv_id":"1912.12887","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deterministic-plus-stochastic-model-of-the","title":"A Deterministic plus Stochastic Model of the Residual Signal for Improved Parametric Speech Synthesis","date":"2019-12-29","arxiv_id":"2001.00842","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-deterministic-plus-stochastic-model-of","title":"The Deterministic plus Stochastic Model of the Residual Signal and its Applications","date":"2019-12-29","arxiv_id":"2001.01000","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-singing-from-speech","title":"Learning Singing From Speech","date":"2019-12-20","arxiv_id":"1912.10128","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-for-whispered-speech","title":"Voice Conversion for Whispered Speech Synthesis","date":"2019-12-11","arxiv_id":"1912.05289","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-neural-vocoding-for-speech","title":"Towards Robust Neural Vocoding for Speech Generation: A Survey","date":"2019-12-05","arxiv_id":"1912.02461","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-speech-synthesis-using-super","title":"High-quality Speech Synthesis Using Super-resolution Mel-Spectrogram","date":"2019-12-03","arxiv_id":"1912.01167","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-prosody-generation-for-speech","title":"Dynamic Prosody Generation for Speech Synthesis using Linguistics-Driven Acoustic Embedding Selection","date":"2019-12-02","arxiv_id":"1912.00955","repositories_listed":0,"syntology":null},{"url":null,"slug":"ji-shi-zhong-wen-yu-yin-he-cheng-xi-tong-real","title":"即時中文語音合成系統 (Real-Time Mandarin Speech Synthesis System)","date":"2019-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-the-next-generation-of-intelligent","title":"Designing the Next Generation of Intelligent Personal Robotic Assistants for the Physically Impaired","date":"2019-11-28","arxiv_id":"1911.12482","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-vaes-and-normalizing-flows-for-one-shot","title":"Using VAEs and Normalizing Flows for One-shot Text-To-Speech Synthesis of Expressive Speech","date":"2019-11-28","arxiv_id":"1911.12760","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-multi-speaker-text-to-speech","title":"Cross-lingual Multi-speaker Text-to-speech Synthesis for Voice Cloning without Using Parallel Corpus for Unseen Speakers","date":"2019-11-26","arxiv_id":"1911.11601","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-sequence-to-sequence-front-end","title":"A unified sequence-to-sequence front-end model for Mandarin text-to-speech synthesis","date":"2019-11-11","arxiv_id":"1911.04111","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-text-to-speech-synthesis-with","title":"Incremental Text-to-Speech Synthesis with Prefix-to-Prefix Framework","date":"2019-11-07","arxiv_id":"1911.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-asvspoof-2019-database","title":"ASVspoof 2019: A large-scale public database of synthesized, converted and replayed speech","date":"2019-11-05","arxiv_id":"1911.01601","repositories_listed":0,"syntology":null},{"url":null,"slug":"replay-spoofing-countermeasure-using","title":"Replay Spoofing Countermeasure Using Autoencoder and Siamese Network on ASVspoof 2019 Challenge","date":"2019-10-29","arxiv_id":"1910.13345","repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-choice-of-probability-distribution","title":"Effect of choice of probability distribution, randomness, and search methods for alignment modeling in sequence-to-sequence text-to-speech synthesis using hard alignment","date":"2019-10-28","arxiv_id":"1910.12383","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-neural-speech-waveform","title":"Transferring neural speech waveform synthesizers to musical instrument sounds generation","date":"2019-10-27","arxiv_id":"1910.12381","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-theory-behind-controllable-expressive","title":"The Theory behind Controllable Expressive Speech Synthesis: a Cross-disciplinary Approach","date":"2019-10-14","arxiv_id":"1910.06234","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-generative-modeling-for","title":"Semi-Supervised Generative Modeling for Controllable Speech Synthesis","date":"2019-10-03","arxiv_id":"1910.01709","repositories_listed":0,"syntology":null},{"url":null,"slug":"ji-shi-zhong-wen-yu-yin-he-cheng-xi-tong-real-1","title":"即時中文語音合成系統(Real-Time Mandarin Speech Synthesis System)","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-nlg-for-speech-synthesis-of","title":"Using NLG for speech synthesis of mathematical sentences","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ying-yong-wen-mo-fen-xi-yu-zhong-ying-jia-za","title":"應用文脈分析於中英夾雜語音合成系統(Linguistic Analysis for English/Mandarin Speech Synthesis System)","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-forcing-for-sequence-to-sequence","title":"Attention Forcing for Sequence-to-sequence Model Training","date":"2019-09-26","arxiv_id":"1909.12289","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-with-augmented-synthesized","title":"Speech Recognition with Augmented Synthesized Speech","date":"2019-09-25","arxiv_id":"1909.11699","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-meta-learning-with-shrinkage","title":"Modular Meta-Learning with Shrinkage","date":"2019-09-12","arxiv_id":"1909.05557","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-long-form-text-to-speech-comparing","title":"Evaluating Long-form Text-to-Speech: Comparing the Ratings of Sentences and Paragraphs","date":"2019-09-09","arxiv_id":"1909.03965","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-harmonic-plus-noise-waveform-model","title":"Neural Harmonic-plus-Noise Waveform Model with Trainable Maximum Voice Frequency for Text-to-Speech Synthesis","date":"2019-08-27","arxiv_id":"1908.10256","repositories_listed":0,"syntology":null}],"record_sha256":"a1cc8017191de235e57f5711d671e222f9b19e877ef8ff2b98a7c81d955aa22f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}