{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-synthesis/papers/3","list_of":"/task/text-to-speech-synthesis","task":"Text-To-Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":332,"counts":{"archive_papers_tagged":332,"with_a_code_link":104,"where_syntology_ran_a_sample":37,"not_listed_spam_title":0,"listed":332,"listed_where_code_ran":37,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":35,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":35,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-synthesis","prev":"/task/text-to-speech-synthesis/papers/2","next":"/task/text-to-speech-synthesis/papers/4","papers":[{"url":null,"slug":"bert-can-he-predict-contrastive-focus","title":"BERT, can HE predict contrastive focus? Predicting and controlling prominence in neural TTS using a language model","date":"2022-07-04","arxiv_id":"2207.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"r-melnet-reduced-mel-spectral-modeling-for","title":"R-MelNet: Reduced Mel-Spectral Modeling for Neural TTS","date":"2022-06-30","arxiv_id":"2206.15276","repositories_listed":0,"syntology":null},{"url":null,"slug":"bu-tts-an-open-source-bilingual-welsh-english","title":"BU-TTS: An Open-Source, Bilingual Welsh-English, Text-to-Speech Corpus","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-transfer-learning-for-urdu-speech","title":"Exploring Transfer Learning for Urdu Speech Synthesis","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-inter-and-intra-speaker-voice","title":"Investigating Inter- and Intra-speaker Voice Conversion using Audiobooks","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recab-vae-gumbel-softmax-variational","title":"ReCAB-VAE: Gumbel-Softmax Variational Inference Based on Analytic Divergence","date":"2022-05-09","arxiv_id":"2205.04104","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-partialspoof-database-and-countermeasures","title":"The PartialSpoof Database and Countermeasures for the Detection of Short Fake Speech Segments Embedded in an Utterance","date":"2022-04-11","arxiv_id":"2204.05177","repositories_listed":0,"syntology":null},{"url":"/paper/somos-the-samsung-open-mos-dataset-for-the","slug":"somos-the-samsung-open-mos-dataset-for-the","title":"SOMOS: The Samsung Open MOS Dataset for the Evaluation of Neural Text-to-Speech Synthesis","date":"2022-04-06","arxiv_id":"2204.03040","repositories_listed":0,"syntology":null},{"url":null,"slug":"vqtts-high-fidelity-text-to-speech-synthesis","title":"VQTTS: High-Fidelity Text-to-Speech Synthesis with Self-Supervised VQ Acoustic Feature","date":"2022-04-02","arxiv_id":"2204.00768","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-syntax-unicode-x2013-prosody-mapping","title":"Applying Syntax$\\unicode{x2013}$Prosody Mapping Hypothesis and Prosodic Well-Formedness Constraints to Neural Sequence-to-Sequence Speech Synthesis","date":"2022-03-29","arxiv_id":"2203.15276","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-duration-modeling-for-end-to","title":"AutoTTS: End-to-End Text-to-Speech Synthesis through Differentiable Duration Modeling","date":"2022-03-21","arxiv_id":"2203.11049","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-free-non-parallel-many-to-many-voice","title":"Text-free non-parallel many-to-many voice conversion using normalising flows","date":"2022-03-15","arxiv_id":"2203.08009","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-performer-score-to-audio-music","title":"Deep Performer: Score-to-Audio Music Performance Synthesis","date":"2022-02-12","arxiv_id":"2202.06034","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-deep-transfer-learning-for-emiot","title":"Multi-Stage Deep Transfer Learning for EmIoT-enabled Human-Computer Interaction","date":"2022-02-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-models-of-text","title":"Transformer-based Models of Text Normalization for Speech Applications","date":"2022-02-01","arxiv_id":"2202.00153","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-multi-style-text-to-speech","title":"Multi-speaker Multi-style Text-to-speech Synthesis With Single-speaker Single-style Training Data Scenarios","date":"2021-12-23","arxiv_id":"2112.12743","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-text-to-speech-with-untranscribed-1","title":"Guided-TTS: A Diffusion Model for Text-to-Speech via Classifier Guidance","date":"2021-11-23","arxiv_id":"2111.11755","repositories_listed":0,"syntology":null},{"url":null,"slug":"environment-aware-text-to-speech-synthesis","title":"Environment Aware Text-to-Speech Synthesis","date":"2021-10-08","arxiv_id":"2110.03887","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-tts-an-end-to-end-speech-synthesis","title":"Prosody-TTS: An end-to-end speech synthesis system with prosody control","date":"2021-10-06","arxiv_id":"2110.02854","repositories_listed":0,"syntology":null},{"url":"/paper/neural-speech-synthesis-in-german","slug":"neural-speech-synthesis-in-german","title":"Neural Speech Synthesis in German","date":"2021-10-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-sequence-to-sequence-networks","title":"Conditioning Sequence-to-sequence Networks with Learned Activations","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-text-to-speech-with-untranscribed","title":"Guided-TTS:Text-to-Speech with Untranscribed Speech","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"low-latency-incremental-text-to-speech","title":"Low-Latency Incremental Text-to-Speech Synthesis with Distilled Context Prediction Network","date":"2021-09-22","arxiv_id":"2109.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-transformer-based-framework-for","title":"A Unified Transformer-based Framework for Duplex Text Normalization","date":"2021-08-23","arxiv_id":"2108.09889","repositories_listed":0,"syntology":null},{"url":null,"slug":"location-location-enhancing-the-evaluation-of","title":"Location, Location: Enhancing the Evaluation of Text-to-Speech Synthesis Using the Rapid Prosody Transcription Paradigm","date":"2021-07-06","arxiv_id":"2107.02527","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-objective-evaluation-of-the-effects-of","title":"An objective evaluation of the effects of recording conditions and speaker characteristics in multi-speaker deep neural speech synthesis","date":"2021-06-03","arxiv_id":"2106.01812","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-verification-derived-loss-and-data","title":"Speaker verification-derived loss and data augmentation for DNN-based multispeaker speech synthesis","date":"2021-06-03","arxiv_id":"2106.01789","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-script-e2e-framework-for-multilingual","title":"Dual Script E2E framework for Multilingual and Code-Switching ASR","date":"2021-06-02","arxiv_id":"2106.01400","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-parsing-based-semantic","title":"Enhancing Word-Level Semantic Representation via Dependency Structure for Expressive Text-to-Speech Synthesis","date":"2021-04-14","arxiv_id":"2104.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"flavored-tacotron-conditional-learning-for","title":"Flavored Tacotron: Conditional Learning for Prosodic-linguistic Features","date":"2021-04-08","arxiv_id":"2104.04050","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-emotional-text-to","title":"Reinforcement Learning for Emotional Text-to-Speech Synthesis with Improved Emotion Discriminability","date":"2021-04-03","arxiv_id":"2104.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"png-bert-augmented-bert-on-phonemes-and","title":"PnG BERT: Augmented BERT on Phonemes and Graphemes for Neural TTS","date":"2021-03-28","arxiv_id":"2103.15060","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-speaker-adaptation-for-text-to","title":"Continual Speaker Adaptation for Text-to-Speech Synthesis","date":"2021-03-26","arxiv_id":"2103.14512","repositories_listed":0,"syntology":null},{"url":null,"slug":"alternate-endings-improving-prosody-for","title":"Alternate Endings: Improving Prosody for Incremental Neural TTS with Predicted Future Text Input","date":"2021-02-19","arxiv_id":"2102.09914","repositories_listed":0,"syntology":null},{"url":null,"slug":"vara-tts-non-autoregressive-text-to-speech","title":"VARA-TTS: Non-Autoregressive Text-to-Speech Synthesis based on Very Deep VAE with Residual Attention","date":"2021-02-12","arxiv_id":"2102.06431","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-cloning-a-multi-speaker-text-to-speech","title":"Voice Cloning: a Multi-Speaker Text-to-Speech Synthesis Approach based on Transfer Learning","date":"2021-02-10","arxiv_id":"2102.05630","repositories_listed":0,"syntology":null},{"url":null,"slug":"triple-m-a-practical-neural-text-to-speech","title":"Triple M: A Practical Text-to-speech Synthesis System With Multi-guidance Attention And Multi-band Multi-time LPCNet","date":"2021-01-30","arxiv_id":"2102.00247","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-wavenet-conditioned-on-vae-latent","title":"Parallel WaveNet conditioned on VAE latent vectors","date":"2020-12-17","arxiv_id":"2012.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-previous-acoustic-context-to-improve","title":"Using previous acoustic context to improve Text-to-Speech synthesis","date":"2020-12-07","arxiv_id":"2012.03763","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-speech-to-speech-translation","title":"Simultaneous Speech-to-Speech Translation System with Neural Incremental ASR, MT, and TTS","date":"2020-11-10","arxiv_id":"2011.04845","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-style-modelling-and-transfer-in","title":"Fine-grained Style Modeling, Transfer and Prediction in Text-to-Speech Synthesis via Phone-Level Content-Style Disentanglement","date":"2020-11-08","arxiv_id":"2011.03943","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-images-for-asr-and-tts-through","title":"Augmenting Images for ASR and TTS through Single-loop and Dual-loop Multimodal Chain Framework","date":"2020-11-04","arxiv_id":"2011.02099","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-machine-speech-chain-towards","title":"Incremental Machine Speech Chain Towards Enabling Listening while Speaking in Real-time","date":"2020-11-04","arxiv_id":"2011.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphspeech-syntax-aware-graph-attention","title":"GraphSpeech: Syntax-Aware Graph Attention Network For Neural Speech Synthesis","date":"2020-10-23","arxiv_id":"2010.12423","repositories_listed":0,"syntology":null},{"url":null,"slug":"grapheme-or-phoneme-an-analysis-of-tacotron-s","title":"An Investigation of the Relation Between Grapheme Embeddings and Pronunciation for Tacotron-based Systems","date":"2020-10-21","arxiv_id":"2010.10694","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-text-to-speech-using-latent","title":"End-to-End Text-to-Speech using Latent Duration based on VQ-VAE","date":"2020-10-19","arxiv_id":"2010.09602","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-arabic-dialect-identification","title":"Automatic Arabic Dialect Identification Systems for Written Texts: A Survey","date":"2020-09-26","arxiv_id":"2009.12622","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-multi-grained-generative-model","title":"Hierarchical Multi-Grained Generative Model for Expressive Speech Synthesis","date":"2020-09-17","arxiv_id":"2009.08474","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-neural-text-to-speech-synthesis","title":"Controllable neural text-to-speech synthesis using intuitive prosodic features","date":"2020-09-14","arxiv_id":"2009.06775","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-the-future-brings-investigating-the","title":"What the Future Brings: Investigating the Impact of Lookahead for Incremental Neural TTS","date":"2020-09-04","arxiv_id":"2009.02035","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-by-cascading-automatic","title":"Voice Conversion by Cascading Automatic Speech Recognition and Text-to-Speech Synthesis with Prosody Transfer","date":"2020-09-03","arxiv_id":"2009.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-text-to-speech-synthesis-using","title":"Multi-speaker Text-to-speech Synthesis Using Deep Gaussian Processes","date":"2020-08-07","arxiv_id":"2008.02950","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalizing-text-using-language-modelling","title":"Normalizing Text using Language Modelling based on Phonetics and String Similarity","date":"2020-06-25","arxiv_id":"2006.14116","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-learning-abilities-on","title":"Investigation of learning abilities on linguistic features in sequence-to-sequence text-to-speech synthesis","date":"2020-05-20","arxiv_id":"2005.10390","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-for-multi-speaker","title":"Semi-supervised Learning for Multi-speaker Text-to-speech Synthesis Using Discrete Speech Representation","date":"2020-05-16","arxiv_id":"2005.08024","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-text-to-speech-synthesis-for-an-under","title":"Neural Text-to-Speech Synthesis for an Under-Resourced Language in a Diglossic Environment: the Case of Gascon Occitan","date":"2020-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"style-variation-as-a-vantage-point-for-code","title":"Style Variation as a Vantage Point for Code-Switching","date":"2020-05-01","arxiv_id":"2005.00458","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-vaes-and-normalizing-flows-for-one-shot","title":"Using VAEs and Normalizing Flows for One-shot Text-To-Speech Synthesis of Expressive Speech","date":"2019-11-28","arxiv_id":"1911.12760","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-multi-speaker-text-to-speech","title":"Cross-lingual Multi-speaker Text-to-speech Synthesis for Voice Cloning without Using Parallel Corpus for Unseen Speakers","date":"2019-11-26","arxiv_id":"1911.11601","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-sequence-to-sequence-front-end","title":"A unified sequence-to-sequence front-end model for Mandarin text-to-speech synthesis","date":"2019-11-11","arxiv_id":"1911.04111","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-text-to-speech-synthesis-with","title":"Incremental Text-to-Speech Synthesis with Prefix-to-Prefix Framework","date":"2019-11-07","arxiv_id":"1911.02750","repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-choice-of-probability-distribution","title":"Effect of choice of probability distribution, randomness, and search methods for alignment modeling in sequence-to-sequence text-to-speech synthesis using hard alignment","date":"2019-10-28","arxiv_id":"1910.12383","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-theory-behind-controllable-expressive","title":"The Theory behind Controllable Expressive Speech Synthesis: a Cross-disciplinary Approach","date":"2019-10-14","arxiv_id":"1910.06234","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-meta-learning-with-shrinkage","title":"Modular Meta-Learning with Shrinkage","date":"2019-09-12","arxiv_id":"1909.05557","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-long-form-text-to-speech-comparing","title":"Evaluating Long-form Text-to-Speech: Comparing the Ratings of Sentences and Paragraphs","date":"2019-09-09","arxiv_id":"1909.03965","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-harmonic-plus-noise-waveform-model","title":"Neural Harmonic-plus-Noise Waveform Model with Trainable Maximum Voice Frequency for Text-to-Speech Synthesis","date":"2019-08-27","arxiv_id":"1908.10256","repositories_listed":0,"syntology":null},{"url":null,"slug":"190600579","title":"Listening while Speaking and Visualizing: Improving ASR through Multimodal Chain","date":"2019-06-03","arxiv_id":"1906.00579","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-models-of-text-normalization-for","title":"Neural Models of Text Normalization for Speech Applications","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-text-normalization-with-subword-units","title":"Neural Text Normalization with Subword Units","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/token-level-ensemble-distillation-for","slug":"token-level-ensemble-distillation-for","title":"Token-Level Ensemble Distillation for Grapheme-to-Phoneme Conversion","date":"2019-04-06","arxiv_id":"1904.03446","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-denoising-by-parametric-resynthesis","title":"Speech denoising by parametric resynthesis","date":"2019-04-02","arxiv_id":"1904.01537","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-network-based-glottal","title":"Generative adversarial network-based glottal waveform model for statistical parametric speech synthesis","date":"2019-03-14","arxiv_id":"1903.05955","repositories_listed":0,"syntology":null},{"url":null,"slug":"atts2s-vc-sequence-to-sequence-voice","title":"AttS2S-VC: Sequence-to-Sequence Voice Conversion with Attention and Context Preservation Mechanisms","date":"2018-11-09","arxiv_id":"1811.04076","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-feedback-loss-in-speech-chain","title":"End-to-End Feedback Loss in Speech Chain Framework via Straight-Through Estimator","date":"2018-10-31","arxiv_id":"1810.13107","repositories_listed":0,"syntology":null},{"url":null,"slug":"waveform-generation-for-text-to-speech","title":"Waveform generation for text-to-speech synthesis using pitch-synchronous multi-scale generative adversarial networks","date":"2018-10-30","arxiv_id":"1810.12598","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaking-style-adaptation-in-text-to-speech","title":"Speaking style adaptation in Text-To-Speech synthesis using Sequence-to-sequence models with attention","date":"2018-10-29","arxiv_id":"1810.12051","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-challenge-set-and-methods-for-noun-verb","title":"A Challenge Set and Methods for Noun-Verb Ambiguity","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-expressive-speaking-style-from","title":"Predicting Expressive Speaking Style From Text In End-To-End Speech Synthesis","date":"2018-08-04","arxiv_id":"1808.01410","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-gan-and-waveform-loss-based","title":"Wasserstein GAN and Waveform Loss-based Acoustic Model Training for Multi-speaker Text-to-Speech Synthesis Systems Using a WaveNet Vocoder","date":"2018-07-31","arxiv_id":"1807.11679","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-development-of-speech-corpora-for","title":"Design and Development of Speech Corpora for Air Traffic Control Training","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-homograph-disambiguation-with","title":"Improving homograph disambiguation with supervised machine learning","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synpaflex-corpus-an-expressive-french","title":"SynPaFlex-Corpus: An Expressive French Audiobooks Corpus dedicated to expressive speech synthesis.","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-independent-raw-waveform-model-for","title":"Speaker-independent raw waveform model for glottal excitation","date":"2018-04-25","arxiv_id":"1804.09593","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-speech-chain-with-one-shot-speaker","title":"Machine Speech Chain with One-shot Speaker Adaptation","date":"2018-03-28","arxiv_id":"1803.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-new-language-and-voice-components","title":"Creating New Language and Voice Components for the Updated MaryTTS Text-to-Speech Synthesis Platform","date":"2017-12-13","arxiv_id":"1712.04787","repositories_listed":0,"syntology":null},{"url":null,"slug":"refer-itts-a-system-for-referring-in-spoken","title":"Refer-iTTS: A System for Referring in Spoken Installments to Objects in Real-World Images","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"listening-while-speaking-speech-chain-by-deep","title":"Listening while Speaking: Speech Chain by Deep Learning","date":"2017-07-16","arxiv_id":"1707.04879","repositories_listed":0,"syntology":null},{"url":null,"slug":"cassandra-a-multipurpose-configurable-voice","title":"CASSANDRA: A multipurpose configurable voice-enabled human-computer-interface","date":"2017-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-syllabification-for-manipuri","title":"Automatic Syllabification for Manipuri language","date":"2016-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dnn-based-speech-synthesis-for-indian","title":"DNN-based Speech Synthesis for Indian Languages from ASCII text","date":"2016-08-18","arxiv_id":"1608.05374","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-taxonomy-of-specific-problem-classes-in","title":"A Taxonomy of Specific Problem Classes in Text-to-Speech Synthesis: Comparing Commercial and Open Source Performance","date":"2016-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"minimally-supervised-number-normalization","title":"Minimally Supervised Number Normalization","date":"2016-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-normalization-and-unit-selection-for-a","title":"Text Normalization and Unit Selection for a Memory Based Non Uniform Unit Selection TTS in Malayalam","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-representation-of-prosody-for","title":"Hierarchical Representation of Prosody for Statistical Speech Synthesis","date":"2015-10-07","arxiv_id":"1510.01949","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-distributed-cloud-based-dialog-system-for","title":"A distributed cloud-based dialog system for conversational application development","date":"2015-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"individuality-preserving-spectrum","title":"Individuality-Preserving Spectrum Modification for Articulation Disorders Using Phone Selective Synthesis","date":"2015-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"which-synthetic-voice-should-i-choose-for-an","title":"Which Synthetic Voice Should I Choose for an Evocative Task?","date":"2015-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-opinions-cross-lingual-opinion","title":"Aligning Opinions: Cross-Lingual Opinion Mining with Dependencies","date":"2015-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"an-in-depth-analysis-of-the-effect-of-text","title":"An In-depth Analysis of the Effect of Text Normalization in Social Media","date":"2015-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"normalization-of-non-standard-words-in","title":"Normalization of Non-Standard Words in Croatian Texts","date":"2015-03-27","arxiv_id":"1503.08167","repositories_listed":0,"syntology":null}],"record_sha256":"3d5ab66fe8d12fc7ca4329e62a42e76f303ce81febdfafae7d32d897072ece50","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}