{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/11","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":13,"rows_per_page":100,"rows":[1001,1100],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/10","next":"/task/speech-synthesis/papers/12","papers":[{"url":null,"slug":"dnn-based-speaker-embedding-using-subjective","title":"DNN-based Speaker Embedding Using Subjective Inter-speaker Similarity for Multi-speaker Modeling in Speech Synthesis","date":"2019-07-19","arxiv_id":"1907.08294","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-end-to-end-speech-synthesis","title":"Multi-Speaker End-to-End Speech Synthesis","date":"2019-07-09","arxiv_id":"1907.04462","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-objective-de-plongements-pour-la","title":"\\'Evaluation objective de plongements pour la synth\\`ese de parole guid\\'ee par r\\'eseaux de neurones (Objective evaluation of embeddings for speech synthesis guided by neural networks)","date":"2019-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-emotional-speech-synthesis-using","title":"End-to-End Emotional Speech Synthesis Using Style Tokens and Semi-Supervised Training","date":"2019-06-26","arxiv_id":"1906.10859","repositories_listed":0,"syntology":null},{"url":"/paper/ruslan-russian-spoken-language-corpus-for","slug":"ruslan-russian-spoken-language-corpus-for","title":"RUSLAN: Russian Spoken Language Corpus for Speech Synthesis","date":"2019-06-26","arxiv_id":"1906.11645","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-speaker-adaptation-method-for","title":"A Unified Speaker Adaptation Method for Speech Synthesis using Transcribed and Untranscribed Speech with Backpropagation","date":"2019-06-18","arxiv_id":"1906.07414","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-transfer-learning-for-end-to-end","title":"Towards Transfer Learning for End-to-End Speech Synthesis from Deep Pre-Trained Language Models","date":"2019-06-17","arxiv_id":"1906.07307","repositories_listed":0,"syntology":null},{"url":null,"slug":"190600579","title":"Listening while Speaking and Visualizing: Improving ASR through Multimodal Chain","date":"2019-06-03","arxiv_id":"1906.00579","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-models-of-text-normalization-for","title":"Neural Models of Text Normalization for Speech Applications","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-text-normalization-with-subword-units","title":"Neural Text Normalization with Subword Units","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"permanent-magnetic-articulograph-pma-vs","title":"Permanent Magnetic Articulograph (PMA) vs Electromagnetic Articulograph (EMA) in Articulation-to-Speech Synthesis for Silent Speech Interface","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-anonymization-using-x-vector-and","title":"Speaker Anonymization Using X-vector and Neural Waveform Models","date":"2019-05-30","arxiv_id":"1905.13561","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-to-video-translation-for-visual-speech","title":"Video-to-Video Translation for Visual Speech Synthesis","date":"2019-05-28","arxiv_id":"1905.12043","repositories_listed":0,"syntology":null},{"url":null,"slug":"eg-gan-cross-language-emotion-gain-synthesis","title":"ET-GAN: Cross-Language Emotion Transfer Based on Cycle-Consistent Generative Adversarial Networks","date":"2019-05-27","arxiv_id":"1905.11173","repositories_listed":0,"syntology":null},{"url":null,"slug":"chive-varying-prosody-in-speech-synthesis","title":"CHiVE: Varying Prosody in Speech Synthesis with a Linguistically Driven Dynamic Hierarchical Conditional Variational Network","date":"2019-05-17","arxiv_id":"1905.07195","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-independent-speech-driven-visual","title":"Speaker-Independent Speech-Driven Visual Speech Synthesis using Domain-Adapted Acoustic Models","date":"2019-05-15","arxiv_id":"1905.06860","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-source-filter-waveform-models-for","title":"Neural source-filter waveform models for statistical parametric speech synthesis","date":"2019-04-27","arxiv_id":"1904.12088","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-variable-algorithms-for-multimodal","title":"Latent Variable Algorithms for Multimodal Learning and Sensor Fusion","date":"2019-04-23","arxiv_id":"1904.10450","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-acoustic-unit-discovery-for","title":"Unsupervised acoustic unit discovery for speech synthesis using discrete latent-variable neural networks","date":"2019-04-16","arxiv_id":"1904.07556","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-high-quality-and-phonetic-balanced-speech","title":"A high quality and phonetic balanced speech corpus for Vietnamese","date":"2019-04-11","arxiv_id":"1904.05569","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-the-eeg-manifold-for","title":"Deep Learning the EEG Manifold for Phonological Categorization from Active Thoughts","date":"2019-04-08","arxiv_id":"1904.04358","repositories_listed":0,"syntology":null},{"url":null,"slug":"speak-your-mind-towards-imagined-speech","title":"SPEAK YOUR MIND! Towards Imagined Speech Recognition With Hierarchical Deep Learning","date":"2019-04-08","arxiv_id":"1904.05746","repositories_listed":0,"syntology":null},{"url":null,"slug":"wavecyclegan2-time-domain-neural-post-filter","title":"WaveCycleGAN2: Time-domain Neural Post-filter for Speech Waveform Generation","date":"2019-04-05","arxiv_id":"1904.02892","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-reference-tacotron-by-intercross","title":"Multi-reference Tacotron by Intercross Training for Style Disentangling,Transfer and Control in Speech Synthesis","date":"2019-04-04","arxiv_id":"1904.02373","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-denoising-by-parametric-resynthesis","title":"Speech denoising by parametric resynthesis","date":"2019-04-02","arxiv_id":"1904.01537","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-training-framework-for-text-to-speech","title":"Joint training framework for text-to-speech and voice conversion using multi-source Tacotron and WaveNet","date":"2019-03-29","arxiv_id":"1903.12389","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-network-based-glottal","title":"Generative adversarial network-based glottal waveform model for statistical parametric speech synthesis","date":"2019-03-14","arxiv_id":"1903.05955","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-text-to-speech-system-with-seq2seq-model","title":"Deep Text-to-Speech System with Seq2Seq Model","date":"2019-03-11","arxiv_id":"1903.07398","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-virtual-doctor-an-interactive-artificial","title":"The Virtual Doctor: An Interactive Artificial Intelligence based on Deep Learning for Non-Invasive Prediction of Diabetes","date":"2019-03-09","arxiv_id":"1903.12069","repositories_listed":0,"syntology":null},{"url":null,"slug":"securing-voice-driven-interfaces-against-fake","title":"Securing Voice-driven Interfaces against Fake (Cloned) Audio Attacks","date":"2019-02-18","arxiv_id":"1902.06782","repositories_listed":0,"syntology":null},{"url":null,"slug":"bytes-are-all-you-need-end-to-end","title":"Bytes are All You Need: End-to-End Multilingual Speech Recognition and Synthesis with Bytes","date":"2018-11-22","arxiv_id":"1811.09021","repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-data-reduction-on-sequence-to","title":"Effect of data reduction on sequence-to-sequence neural TTS","date":"2018-11-15","arxiv_id":"1811.06315","repositories_listed":0,"syntology":null},{"url":null,"slug":"atts2s-vc-sequence-to-sequence-voice","title":"AttS2S-VC: Sequence-to-Sequence Voice Conversion with Attention and Context Preservation Mechanisms","date":"2018-11-09","arxiv_id":"1811.04076","repositories_listed":0,"syntology":null},{"url":null,"slug":"excitnet-vocoder-a-neural-excitation-model","title":"ExcitNet vocoder: A neural excitation model for parametric speech synthesis systems","date":"2018-11-09","arxiv_id":"1811.04769","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-adaptive-neural-vocoders-for","title":"Speaker-adaptive neural vocoders for parametric speech synthesis systems","date":"2018-11-08","arxiv_id":"1811.03311","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-context-features-hidden-in-end","title":"Investigating context features hidden in End-to-End TTS","date":"2018-11-04","arxiv_id":"1811.01376","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-feedback-loss-in-speech-chain","title":"End-to-End Feedback Loss in Speech Chain Framework via Straight-Through Estimator","date":"2018-10-31","arxiv_id":"1810.13107","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-correlated-speaker-and-noise","title":"Disentangling Correlated Speaker and Noise for Speech Synthesis via Data Augmentation and Adversarial Factorization","date":"2018-10-30","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-make-someone-speak-a-language-that","title":"How to make someone speak a language that they don't know.","date":"2018-10-30","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"waveform-generation-for-text-to-speech","title":"Waveform generation for text-to-speech synthesis using pitch-synchronous multi-scale generative adversarial networks","date":"2018-10-30","arxiv_id":"1810.12598","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-source-filter-based-waveform-model-for","title":"Neural source-filter-based waveform model for statistical parametric speech synthesis","date":"2018-10-29","arxiv_id":"1810.11946","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaking-style-adaptation-in-text-to-speech","title":"Speaking style adaptation in Text-To-Speech synthesis using Sequence-to-sequence models with attention","date":"2018-10-29","arxiv_id":"1810.12051","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-challenge-set-and-methods-for-noun-verb","title":"A Challenge Set and Methods for Noun-Verb Ambiguity","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wavecyclegan-synthetic-to-natural-speech","title":"WaveCycleGAN: Synthetic-to-natural speech waveform conversion using cycle-consistent adversarial networks","date":"2018-09-25","arxiv_id":"1809.10288","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindi-english-code-switching-speech-corpus","title":"Hindi-English Code-Switching Speech Corpus","date":"2018-09-24","arxiv_id":"1810.00662","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-attention-linguistic-acoustic-decoder","title":"Self-Attention Linguistic-Acoustic Decoder","date":"2018-08-31","arxiv_id":"1808.10678","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-training-for-improving-data","title":"Semi-Supervised Training for Improving Data Efficiency in End-to-End Speech Synthesis","date":"2018-08-30","arxiv_id":"1808.10128","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-spectrogram-inversion-using-multi-head","title":"Fast Spectrogram Inversion using Multi-head Convolutional Neural Networks","date":"2018-08-20","arxiv_id":"1808.06719","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-speech-synthesis-architecture-for","title":"Multimodal speech synthesis architecture for unsupervised speaker adaptation","date":"2018-08-20","arxiv_id":"1808.06288","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-expressive-speaking-style-from","title":"Predicting Expressive Speaking Style From Text In End-To-End Speech Synthesis","date":"2018-08-04","arxiv_id":"1808.01410","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-accuracy-of-pitch-accent","title":"Investigating accuracy of pitch-accent annotations in neural network-based speech synthesis and denoising effects","date":"2018-08-02","arxiv_id":"1808.00665","repositories_listed":0,"syntology":null},{"url":null,"slug":"indigenous-language-technologies-in-canada","title":"Indigenous language technologies in Canada: Assessment, challenges, and successes","date":"2018-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-and-bias-codes-for-modeling-speaker","title":"Scaling and bias codes for modeling speaker-adaptive DNN-based speech synthesis systems","date":"2018-07-31","arxiv_id":"1807.11632","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-gan-and-waveform-loss-based","title":"Wasserstein GAN and Waveform Loss-based Acoustic Model Training for Multi-speaker Text-to-Speech Synthesis Systems Using a WaveNet Vocoder","date":"2018-07-31","arxiv_id":"1807.11679","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-encoder-decoder-models-for-unsupervised","title":"Deep Encoder-Decoder Models for Unsupervised Learning of Controllable Speech Synthesis","date":"2018-07-30","arxiv_id":"1807.11470","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysing-shortcomings-of-statistical","title":"Analysing Shortcomings of Statistical Parametric Speech Synthesis","date":"2018-07-28","arxiv_id":"1807.10941","repositories_listed":0,"syntology":null},{"url":null,"slug":"forward-attention-in-sequence-to-sequence","title":"Forward Attention in Sequence-to-sequence Acoustic Modelling for Speech Synthesis","date":"2018-07-18","arxiv_id":"1807.06736","repositories_listed":0,"syntology":null},{"url":null,"slug":"syllabification-by-phone-categorization","title":"Syllabification by Phone Categorization","date":"2018-07-15","arxiv_id":"1807.05518","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-the-effect-of-emotional-speech","title":"An Analysis of the Effect of Emotional Speech Synthesis on Non-Task-Oriented Dialogue System","date":"2018-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"emphasis-an-emotional-phoneme-based-acoustic","title":"EMPHASIS: An Emotional Phoneme-based Acoustic Model for Speech Synthesis System","date":"2018-06-26","arxiv_id":"1806.09276","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-wavenet-a-multi-task-generative","title":"Multi-task WaveNet: A Multi-task Generative Model for Statistical Parametric Speech Synthesis without Fundamental Frequency Conditions","date":"2018-06-22","arxiv_id":"1806.08619","repositories_listed":0,"syntology":null},{"url":null,"slug":"asr-based-features-for-emotion-recognition-a","title":"ASR-based Features for Emotion Recognition: A Transfer Learning Approach","date":"2018-05-23","arxiv_id":"1805.09197","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-regression-model-of-recurrent-deep-neural","title":"A Regression Model of Recurrent Deep Neural Networks for Noise Robust Estimation of the Fundamental Frequency Contour of Speech","date":"2018-05-08","arxiv_id":"1805.02958","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-open-javanese-and-sundanese-corpora","title":"Building Open Javanese and Sundanese Corpora for Multilingual Text-to-Speech","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"construction-of-english-french-multimodal","title":"Construction of English-French Multimodal Affective Conversational Corpus from TV Dramas","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cpjd-corpus-crowdsourced-parallel-speech","title":"CPJD Corpus: Crowdsourced Parallel Speech Corpus of Japanese Dialects","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"design-and-development-of-speech-corpora-for","title":"Design and Development of Speech Corpora for Air Traffic Control Training","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"from-asolved-problemsa-to-new-challenges-a","title":"From `Solved Problems' to New Challenges: A Report on LDC Activities","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-homograph-disambiguation-with","title":"Improving homograph disambiguation with supervised machine learning","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"literality-and-cognitive-effort-japanese-and","title":"Literality and cognitive effort: Japanese and Spanish","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synpaflex-corpus-an-expressive-french","title":"SynPaFlex-Corpus: An Expressive French Audiobooks Corpus dedicated to expressive speech synthesis.","date":"2018-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-independent-raw-waveform-model-for","title":"Speaker-independent raw waveform model for glottal excitation","date":"2018-04-25","arxiv_id":"1804.09593","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-recent-waveform-generation","title":"A comparison of recent waveform generation and acoustic modeling methods for neural-network-based speech synthesis","date":"2018-04-07","arxiv_id":"1804.02549","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-speech-synthesis-via-modeling","title":"Expressive Speech Synthesis via Modeling Expressions with Variational Autoencoder","date":"2018-04-06","arxiv_id":"1804.02135","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-nonparallel-voice-conversion","title":"High-quality nonparallel voice conversion based on cycle-consistent adversarial network","date":"2018-04-02","arxiv_id":"1804.00425","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-speech-chain-with-one-shot-speaker","title":"Machine Speech Chain with One-shot Speaker Adaptation","date":"2018-03-28","arxiv_id":"1803.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-steal-your-vocal-identity-from-the","title":"Can we steal your vocal identity from the Internet?: Initial investigation of cloning Obama's voice using GAN, WaveNet and low-quality found data","date":"2018-03-02","arxiv_id":"1803.00860","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-feed-forward-sequential-memory-networks","title":"Deep Feed-forward Sequential Memory Networks for Speech Synthesis","date":"2018-02-26","arxiv_id":"1802.09194","repositories_listed":0,"syntology":null},{"url":null,"slug":"fitting-new-speakers-based-on-a-short","title":"Fitting New Speakers Based on a Short Untranscribed Sample","date":"2018-02-20","arxiv_id":"1802.06984","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybridnet-a-hybrid-neural-architecture-to","title":"HybridNet: A Hybrid Neural Architecture to Speed-up Autoregressive Models","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-driven-generative-adversarial-networks","title":"POLICY DRIVEN GENERATIVE ADVERSARIAL NETWORKS FOR ACCENTED SPEECH GENERATION","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pyiwn-a-python-based-api-to-access-indian","title":"pyiwn: A Python based API to access Indian Language WordNets","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-audio-for-hindi-wordnet","title":"Synthesizing Audio for Hindi WordNet","date":"2018-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/merging-k-means-with-hierarchical-clustering","slug":"merging-k-means-with-hierarchical-clustering","title":"Merging $K$-means with hierarchical clustering for identifying general-shaped groups","date":"2017-12-23","arxiv_id":"1712.08786","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-new-language-and-voice-components","title":"Creating New Language and Voice Components for the Updated MaryTTS Text-to-Speech Synthesis Platform","date":"2017-12-13","arxiv_id":"1712.04787","repositories_listed":0,"syntology":null},{"url":null,"slug":"aa-ao14eccc2e-a1eae3ac3cac-c-a-preliminary","title":"完全基於類神經網路之語音合成系統初步研究 (A Preliminary Study on Fully Neural Network-based Speech Synthesis System) [In Chinese]","date":"2017-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sut-system-description-for-anti-spoofing-2017","title":"SUT System Description for Anti-Spoofing 2017 Challenge","date":"2017-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"uncovering-latent-style-factors-for","title":"Uncovering Latent Style Factors for Expressive Speech Synthesis","date":"2017-11-01","arxiv_id":"1711.00520","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-accurate-decision-trees-for-natural","title":"Fast and Accurate Decision Trees for Natural Language Processing Tasks","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lexicon-for-natural-language-generation-in","title":"Lexicon for Natural Language Generation in Spanish Adapted to Alternative and Augmentative Communication","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"refer-itts-a-system-for-referring-in-spoken","title":"Refer-iTTS: A System for Referring in Spoken Installments to Objects in Real-World Images","date":"2017-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"listening-while-speaking-speech-chain-by-deep","title":"Listening while Speaking: Speech Chain by Deep Learning","date":"2017-07-16","arxiv_id":"1707.04879","repositories_listed":0,"syntology":null},{"url":null,"slug":"hidden-markov-model-based-speech-enhancement","title":"Hidden-Markov-Model Based Speech Enhancement","date":"2017-07-04","arxiv_id":"1707.01090","repositories_listed":0,"syntology":null},{"url":null,"slug":"pydial-a-multi-domain-statistical-dialogue","title":"PyDial: A Multi-domain Statistical Dialogue System Toolkit","date":"2017-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-variational-em-method-for-pole-zero","title":"A Variational EM Method for Pole-Zero Modeling of Speech with Mixed Block Sparse and Gaussian Excitation","date":"2017-06-24","arxiv_id":"1706.07927","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-probe-therefore-i-am-designing-a-virtual","title":"I Probe, Therefore I Am: Designing a Virtual Journalist with Human Emotions","date":"2017-05-18","arxiv_id":"1705.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-phonemes-using-finte-state-methods","title":"Aligning phonemes using finte-state methods","date":"2017-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-and-using-language-resources-and","title":"Building and using language resources and infrastructure to develop e-learning programs for a minority language","date":"2017-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sampling-based-speech-parameter-generation","title":"Sampling-based speech parameter generation using moment-matching networks","date":"2017-04-12","arxiv_id":"1704.03626","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-using-sequence-to-sequence","title":"Voice Conversion Using Sequence-to-Sequence Learning of Context Posterior Probabilities","date":"2017-04-10","arxiv_id":"1704.02360","repositories_listed":0,"syntology":null}],"record_sha256":"848c563805a467d81e62ca053a53a840300c6d73cc026eb3cf24ab66ce304a31","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}