{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech/papers/13","list_of":"/task/text-to-speech","task":"Text to Speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":15,"rows_per_page":100,"rows":[1201,1300],"of":1419,"counts":{"archive_papers_tagged":1419,"with_a_code_link":399,"where_syntology_ran_a_sample":108,"not_listed_spam_title":0,"listed":1419,"listed_where_code_ran":108,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech","prev":"/task/text-to-speech/papers/12","next":"/task/text-to-speech/papers/14","papers":[{"url":null,"slug":"incremental-machine-speech-chain-towards","title":"Incremental Machine Speech Chain Towards Enabling Listening while Speaking in Real-time","date":"2020-11-04","arxiv_id":"2011.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodic-representation-learning-and","title":"Prosodic Representation Learning and Contextual Sampling for Neural Text-to-Speech","date":"2020-11-04","arxiv_id":"2011.02252","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-wake-word-detection-with-synthesized","title":"Training Wake Word Detection with Synthesized Speech Data on Confusion Words","date":"2020-11-03","arxiv_id":"2011.01460","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptually-guided-end-to-end-text-to-speech","title":"Learning to Maximize Speech Quality Directly Using MOS Prediction for Neural Text-to-Speech","date":"2020-11-02","arxiv_id":"2011.01174","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-explanations-and-demonstrations","title":"Learning from Explanations and Demonstrations: A Pilot Study","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"devicetts-a-small-footprint-fast-stable","title":"DeviceTTS: A Small-Footprint, Fast, Stable Network for On-Device Text-to-Speech","date":"2020-10-29","arxiv_id":"2010.15311","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-decoder-masking-for-transformer","title":"Effective Decoder Masking for Transformer Based End-to-End Speech Recognition","date":"2020-10-27","arxiv_id":"2010.14764","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-waveform-synthesis-based-on","title":"Parallel waveform synthesis based on generative adversarial networks with voicing-aware conditional discriminators","date":"2020-10-27","arxiv_id":"2010.14151","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotion-controllable-speech-synthesis-using","title":"Emotion controllable speech synthesis using emotion-unlabeled dataset with the assistance of cross-domain speech emotion recognition","date":"2020-10-26","arxiv_id":"2010.13350","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphspeech-syntax-aware-graph-attention","title":"GraphSpeech: Syntax-Aware Graph Attention Network For Neural Speech Synthesis","date":"2020-10-23","arxiv_id":"2010.12423","repositories_listed":0,"syntology":null},{"url":null,"slug":"nu-gan-high-resolution-neural-upsampling-with","title":"NU-GAN: High resolution neural upsampling with GAN","date":"2020-10-22","arxiv_id":"2010.11362","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ntu-aisg-text-to-speech-system-for","title":"The NTU-AISG Text-to-speech System for Blizzard Challenge 2020","date":"2020-10-22","arxiv_id":"2010.11489","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mask-based-model-for-mandarin-chinese","title":"A Mask-based Model for Mandarin Chinese Polyphone Disambiguation","date":"2020-10-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"grapheme-or-phoneme-an-analysis-of-tacotron-s","title":"An Investigation of the Relation Between Grapheme Embeddings and Pronunciation for Tacotron-based Systems","date":"2020-10-21","arxiv_id":"2010.10694","repositories_listed":0,"syntology":null},{"url":null,"slug":"replacing-human-audio-with-synthetic-audio","title":"Replacing Human Audio with Synthetic Audio for On-device Unspoken Punctuation Prediction","date":"2020-10-20","arxiv_id":"2010.10203","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-text-to-speech-using-latent","title":"End-to-End Text-to-Speech using Latent Duration based on VQ-VAE","date":"2020-10-19","arxiv_id":"2010.09602","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-low-resource-code-switched-asr","title":"Improving Low Resource Code-switched ASR using Augmented Code-switched TTS","date":"2020-10-12","arxiv_id":"2010.05549","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-linguistic-embedding-for-cross-lingual","title":"Latent linguistic embedding for cross-lingual text-to-speech and voice conversion","date":"2020-10-08","arxiv_id":"2010.03717","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-unpaired-text-data-for-training","title":"Leveraging Unpaired Text Data for Training End-to-End Speech-to-Intent Systems","date":"2020-10-08","arxiv_id":"2010.04284","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-speech-synthesis-for-estonian","title":"Neural Speech Synthesis for Estonian","date":"2020-10-06","arxiv_id":"2010.02636","repositories_listed":0,"syntology":null},{"url":null,"slug":"compress-polyphone-pronunciation-prediction","title":"Compress Polyphone Pronunciation Prediction Model with Shared Labels","date":"2020-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-arabic-dialect-identification","title":"Automatic Arabic Dialect Identification Systems for Written Texts: A Survey","date":"2020-09-26","arxiv_id":"2009.12622","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-multi-grained-generative-model","title":"Hierarchical Multi-Grained Generative Model for Expressive Speech Synthesis","date":"2020-09-17","arxiv_id":"2009.08474","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-neural-text-to-speech-synthesis","title":"Controllable neural text-to-speech synthesis using intuitive prosodic features","date":"2020-09-14","arxiv_id":"2009.06775","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-the-future-brings-investigating-the","title":"What the Future Brings: Investigating the Impact of Lookahead for Incremental Neural TTS","date":"2020-09-04","arxiv_id":"2009.02035","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-by-cascading-automatic","title":"Voice Conversion by Cascading Automatic Speech Recognition and Text-to-Speech Synthesis with Prosody Transfer","date":"2020-09-03","arxiv_id":"2009.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"textual-echo-cancellation","title":"Textual Echo Cancellation","date":"2020-08-13","arxiv_id":"2008.06006","repositories_listed":0,"syntology":null},{"url":null,"slug":"bunched-lpcnet-vocoder-for-low-cost-neural","title":"Bunched LPCNet : Vocoder for Low-cost Neural Text-To-Speech Systems","date":"2020-08-11","arxiv_id":"2008.04574","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-learning-for-sequence-to","title":"Unsupervised Learning For Sequence-to-sequence Text-to-speech For Low-resource Languages","date":"2020-08-11","arxiv_id":"2008.04549","repositories_listed":0,"syntology":null},{"url":null,"slug":"lrspeech-extremely-low-resource-speech","title":"LRSpeech: Extremely Low-Resource Speech Synthesis and Recognition","date":"2020-08-09","arxiv_id":"2008.03687","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-text-to-speech-for-neural","title":"Incremental Text to Speech for Neural Sequence-to-Sequence Models using Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03096","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-text-to-speech-synthesis-using","title":"Multi-speaker Text-to-speech Synthesis Using Deep Gaussian Processes","date":"2020-08-07","arxiv_id":"2008.02950","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-rnn-t-models-surpassing-high","title":"Developing RNN-T Models Surpassing High-Performance Hybrid Models with Customization Capability","date":"2020-07-30","arxiv_id":"2007.15188","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-transfer-learning-end-to-end-arabictext-to","title":"A Transfer Learning End-to-End ArabicText-To-Speech (TTS) Deep Architecture","date":"2020-07-22","arxiv_id":"2007.11541","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalizing-text-using-language-modelling","title":"Normalizing Text using Language Modelling based on Phonetics and String Similarity","date":"2020-06-25","arxiv_id":"2006.14116","repositories_listed":0,"syntology":null},{"url":null,"slug":"generic-indic-text-to-speech-synthesisers","title":"Generic Indic Text-to-speech Synthesisers with Rapid Adaptation in an End-to-end Framework","date":"2020-06-12","arxiv_id":"2006.06971","repositories_listed":0,"syntology":null},{"url":null,"slug":"nautilus-a-versatile-voice-cloning-system","title":"NAUTILUS: a Versatile Voice Cloning System","date":"2020-05-22","arxiv_id":"2005.11004","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-multispeaker-text-to-speech","title":"Cross-lingual Multispeaker Text-to-Speech under Limited-Data Scenario","date":"2020-05-21","arxiv_id":"2005.10441","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-learning-abilities-on","title":"Investigation of learning abilities on linguistic features in sequence-to-sequence text-to-speech synthesis","date":"2020-05-20","arxiv_id":"2005.10390","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accent-conversion-with-reference","title":"Improving Accent Conversion with Reference Encoder and End-To-End Text-To-Speech","date":"2020-05-19","arxiv_id":"2005.09271","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-and-data-driven-amplitude-spectrum","title":"Knowledge-and-Data-Driven Amplitude Spectrum Prediction for Hierarchical Neural Vocoders","date":"2020-05-18","arxiv_id":"2004.07832","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-for-multi-speaker","title":"Semi-supervised Learning for Multi-speaker Text-to-speech Synthesis Using Discrete Speech Representation","date":"2020-05-16","arxiv_id":"2005.08024","repositories_listed":0,"syntology":null},{"url":null,"slug":"jdi-t-jointly-trained-duration-informed","title":"JDI-T: Jointly trained Duration Informed Transformer for Text-To-Speech without Explicit Alignment","date":"2020-05-15","arxiv_id":"2005.07799","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-do-not-need-more-data-improving-end-to","title":"You Do Not Need More Data: Improving End-To-End Speech Recognition by Text-To-Speech Data Augmentation","date":"2020-05-14","arxiv_id":"2005.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"adadurian-few-shot-adaptation-for-neural-text","title":"AdaDurIAN: Few-shot Adaptation for Neural Text-to-Speech with DurIAN","date":"2020-05-12","arxiv_id":"2005.05642","repositories_listed":0,"syntology":null},{"url":null,"slug":"discretalk-text-to-speech-as-a-machine","title":"DiscreTalk: Text-to-Speech as a Machine Translation Problem","date":"2020-05-12","arxiv_id":"2005.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"burmese-speech-corpus-finite-state-text","title":"Burmese Speech Corpus, Finite-State Text Normalization and Pronunciation Grammars with an Application to Text-to-Speech","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"corpus-generation-for-voice-command-in-smart","title":"Corpus Generation for Voice Command in Smart Home and the Effect of Speech Synthesis on End-to-End SLU","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"crowdsourcing-latin-american-spanish-for-low","title":"Crowdsourcing Latin American Spanish for Low-Resource Text-to-Speech","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"development-and-evaluation-of-speech","title":"Development and Evaluation of Speech Synthesis Corpora for Latvian","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"indicspeech-text-to-speech-corpus-for-indian","title":"IndicSpeech: Text-to-Speech Corpus for Indian Languages","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-text-to-speech-synthesis-for-an-under","title":"Neural Text-to-Speech Synthesis for an Under-Resourced Language in a Diglossic Environment: the Case of Gascon Occitan","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/open-source-high-quality-speech-datasets-for","slug":"open-source-high-quality-speech-datasets-for","title":"Open-Source High Quality Speech Datasets for Basque, Catalan and Galician","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"open-source-multi-speaker-speech-corpora-for","title":"Open-source Multi-speaker Speech Corpora for Building Gujarati, Kannada, Malayalam, Marathi, Tamil and Telugu Speech Synthesis Systems","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"style-variation-as-a-vantage-point-for-code","title":"Style Variation as a Vantage Point for Code-Switching","date":"2020-05-01","arxiv_id":"2005.00458","repositories_listed":0,"syntology":null},{"url":null,"slug":"copycat-many-to-many-fine-grained-prosody","title":"CopyCat: Many-to-Many Fine-Grained Prosody Transfer for Neural Text-to-Speech","date":"2020-04-30","arxiv_id":"2004.14617","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-non-autoregressive-model-for","title":"A Study of Non-autoregressive Model for Sequence Generation","date":"2020-04-22","arxiv_id":"2004.10454","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-processing-for-optimizing-naturalness-of","title":"Data Processing for Optimizing Naturalness of Vietnamese Text-to-speech System","date":"2020-04-20","arxiv_id":"2004.09607","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-multilingual-voices-using-speaker","title":"Generating Multilingual Voices Using Speaker Space Translation Based on Bilingual Speaker Data","date":"2020-04-10","arxiv_id":"2004.04972","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multilingual-frontend-for-tts","title":"Scalable Multilingual Frontend for TTS","date":"2020-04-10","arxiv_id":"2004.04934","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-readability-for-automatic-speech","title":"Improving Readability for Automatic Speech Recognition Transcription","date":"2020-04-09","arxiv_id":"2004.04438","repositories_listed":0,"syntology":null},{"url":null,"slug":"tatistical-context-dependent-units-boundary","title":"Statistical Context-Dependent Units Boundary Correction for Corpus-based Unit-Selection Text-to-Speech","date":"2020-03-05","arxiv_id":"2003.02837","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphtts-graph-to-sequence-modelling-in","title":"GraphTTS: graph-to-sequence modelling in neural text-to-speech","date":"2020-03-04","arxiv_id":"2003.01924","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-hierarchical-fine-grained-prosody","title":"Fully-hierarchical fine-grained prosody modeling for interpretable speech synthesis","date":"2020-02-06","arxiv_id":"2002.03785","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-diverse-and-natural-text-to-speech","title":"Generating diverse and natural text-to-speech samples using a quantized fine-grained VAE and auto-regressive prosody prior","date":"2020-02-06","arxiv_id":"2002.03788","repositories_listed":0,"syntology":null},{"url":null,"slug":"boffin-tts-few-shot-speaker-adaptation-by","title":"BOFFIN TTS: Few-Shot Speaker Adaptation by Bayesian Optimization","date":"2020-02-04","arxiv_id":"2002.01953","repositories_listed":0,"syntology":null},{"url":"/paper/wavetts-tacotron-based-tts-with-joint-time","slug":"wavetts-tacotron-based-tts-with-joint-time","title":"WaveTTS: Tacotron-based TTS with Joint Time-Frequency Domain Loss","date":"2020-02-02","arxiv_id":"2002.00417","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-speech-to-speech-translation-to","title":"From Speech-to-Speech Translation to Automatic Dubbing","date":"2020-01-19","arxiv_id":"2001.06785","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-neural-text-to-speech-1","title":"Parallel Neural Text-to-Speech","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-summarizer-for-blind-people","title":"Smart Summarizer for Blind People","date":"2020-01-01","arxiv_id":"2001.00575","repositories_listed":0,"syntology":null},{"url":null,"slug":"singing-synthesis-with-a-little-help-from-my","title":"Singing Synthesis: with a little help from my attention","date":"2019-12-12","arxiv_id":"1912.05881","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-neural-vocoding-for-speech","title":"Towards Robust Neural Vocoding for Speech Generation: A Survey","date":"2019-12-05","arxiv_id":"1912.02461","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-prosody-generation-for-speech","title":"Dynamic Prosody Generation for Speech Synthesis using Linguistics-Driven Acoustic Embedding Selection","date":"2019-12-02","arxiv_id":"1912.00955","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-vaes-and-normalizing-flows-for-one-shot","title":"Using VAEs and Normalizing Flows for One-shot Text-To-Speech Synthesis of Expressive Speech","date":"2019-11-28","arxiv_id":"1911.12760","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-multi-speaker-text-to-speech","title":"Cross-lingual Multi-speaker Text-to-speech Synthesis for Voice Cloning without Using Parallel Corpus for Unseen Speakers","date":"2019-11-26","arxiv_id":"1911.11601","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-transfer-in-neural-text-to-speech","title":"Prosody Transfer in Neural Text to Speech Using Global Pitch and Loudness Features","date":"2019-11-21","arxiv_id":"1911.09645","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-sequence-to-sequence-front-end","title":"A unified sequence-to-sequence front-end model for Mandarin text-to-speech synthesis","date":"2019-11-11","arxiv_id":"1911.04111","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-text-to-speech-synthesis-with","title":"Incremental Text-to-Speech Synthesis with Prefix-to-Prefix Framework","date":"2019-11-07","arxiv_id":"1911.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-student-training-for-robust-tacotron","title":"Teacher-Student Training for Robust Tacotron-based TTS","date":"2019-11-07","arxiv_id":"1911.02839","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-system-for-diacritizing-four-varieties-of","title":"A System for Diacritizing Four Varieties of Arabic","date":"2019-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-choice-of-probability-distribution","title":"Effect of choice of probability distribution, randomness, and search methods for alignment modeling in sequence-to-sequence text-to-speech synthesis using hard alignment","date":"2019-10-28","arxiv_id":"1910.12383","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-pre-traing-for-sequence-to","title":"Unsupervised pre-training for sequence to sequence speech recognition","date":"2019-10-28","arxiv_id":"1910.12418","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-reference-neural-tts-stylization-with","title":"Multi-Reference Neural TTS Stylization with Adversarial Cycle Consistency","date":"2019-10-25","arxiv_id":"1910.11958","repositories_listed":0,"syntology":null},{"url":null,"slug":"g2g-tts-driven-pronunciation-learning-for","title":"G2G: TTS-Driven Pronunciation Learning for Graphemic Hybrid ASR","date":"2019-10-22","arxiv_id":"1910.12612","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-theory-behind-controllable-expressive","title":"The Theory behind Controllable Expressive Speech Synthesis: a Cross-disciplinary Approach","date":"2019-10-14","arxiv_id":"1910.06234","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-generative-modeling-for","title":"Semi-Supervised Generative Modeling for Controllable Speech Synthesis","date":"2019-10-03","arxiv_id":"1910.01709","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-non-parallel-voice-conversion","title":"Bootstrapping non-parallel voice conversion from speaker-adaptive text-to-speech","date":"2019-09-14","arxiv_id":"1909.06532","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-meta-learning-with-shrinkage","title":"Modular Meta-Learning with Shrinkage","date":"2019-09-12","arxiv_id":"1909.05557","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-long-form-text-to-speech-comparing","title":"Evaluating Long-form Text-to-Speech: Comparing the Ratings of Sentences and Paragraphs","date":"2019-09-09","arxiv_id":"1909.03965","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-based-modeling-of-phonetic","title":"Neural Network-Based Modeling of Phonetic Durations","date":"2019-09-06","arxiv_id":"1909.03030","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-large-scale-user-study-of-an-alexa-prize","title":"A Large-Scale User Study of an Alexa Prize Chatbot: Effect of TTS Dynamism on Perceived Quality of Social Dialog","date":"2019-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"initial-investigation-of-an-encoder-decoder","title":"Initial investigation of an encoder-decoder end-to-end TTS framework using marginalization of monotonic hard latent alignments","date":"2019-08-30","arxiv_id":"1908.11535","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-harmonic-plus-noise-waveform-model","title":"Neural Harmonic-plus-Noise Waveform Model with Trainable Maximum Voice Frequency for Text-to-Speech Synthesis","date":"2019-08-27","arxiv_id":"1908.10256","repositories_listed":0,"syntology":null},{"url":null,"slug":"190807590","title":"From Text to Sound: A Preliminary Study on Retrieving Sound Effects to Radio Stories","date":"2019-08-20","arxiv_id":"1908.07590","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-sequence-to-sequence-voice","title":"Hierarchical Sequence to Sequence Voice Conversion with Limited Data","date":"2019-07-15","arxiv_id":"1907.07769","repositories_listed":0,"syntology":null},{"url":null,"slug":"m3d-gan-multi-modal-multi-domain-translation","title":"M3D-GAN: Multi-Modal Multi-Domain Translation with Universal Attention","date":"2019-07-09","arxiv_id":"1907.04378","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-methodology-for-controlling-the-emotional","title":"A Methodology for Controlling the Emotional Expressiveness in Synthetic Speech -- a Deep Learning approach","date":"2019-07-05","arxiv_id":"1907.02784","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-approach-to-ocr-using-image","title":"A Novel Approach to OCR using Image Recognition based Classification for Ancient Tamil Inscriptions in Temples","date":"2019-07-04","arxiv_id":"1907.04917","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-robust-prosody-transfer-for","title":"Fine-grained robust prosody transfer for single-speaker neural text-to-speech","date":"2019-07-04","arxiv_id":"1907.02479","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyphone-disambiguation-for-mandarin-chinese","title":"Polyphone Disambiguation for Mandarin Chinese Using Conditional Neural Network with Multi-level Embedding Features","date":"2019-07-03","arxiv_id":"1907.01749","repositories_listed":0,"syntology":null}],"record_sha256":"bfe08480e644f1dcd6b169668d2261419ade9901738bd8a17fafe990fc1b3b7b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}