{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-synthesis/papers/8","list_of":"/task/speech-synthesis","task":"Speech Synthesis","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":13,"rows_per_page":100,"rows":[701,800],"of":1249,"counts":{"archive_papers_tagged":1249,"with_a_code_link":366,"where_syntology_ran_a_sample":101,"not_listed_spam_title":0,"listed":1249,"listed_where_code_ran":101,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-synthesis","prev":"/task/speech-synthesis/papers/7","next":"/task/speech-synthesis/papers/9","papers":[{"url":null,"slug":"any-speaker-adaptive-text-to-speech-synthesis","title":"Grad-StyleSpeech: Any-speaker Adaptive Text-to-Speech Synthesis with Diffusion Models","date":"2022-11-17","arxiv_id":"2211.09383","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-anti-spoofing-using-a-simple-attention","title":"Audio Anti-spoofing Using a Simple Attention Module and Joint Optimization Based on Additive Angular Margin Loss and Meta-learning","date":"2022-11-17","arxiv_id":"2211.09898","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-potential-of-neural-speech-synthesis","title":"The Potential of Neural Speech Synthesis-based Data Augmentation for Personalized Speech Enhancement","date":"2022-11-14","arxiv_id":"2211.07493","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-emotional-intensity-controllable","title":"Semi-supervised learning for continuous emotional intensity controllable speech synthesis with disentangled representations","date":"2022-11-11","arxiv_id":"2211.06160","repositories_listed":0,"syntology":null},{"url":null,"slug":"deliberation-networks-and-how-to-train-them","title":"Deliberation Networks and How to Train Them","date":"2022-11-06","arxiv_id":"2211.03217","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-phoneme-level-prosody-latents","title":"Predicting phoneme-level prosody latents using AR and flow-based Prior Networks for expressive speech synthesis","date":"2022-11-02","arxiv_id":"2211.01327","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-study-on-mandarin-hakka-neural","title":"A Preliminary Study on Mandarin-Hakka neural machine translation using small-sized data","date":"2022-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-mandarin-english-code","title":"Development of Mandarin-English code-switching speech synthesis system","date":"2022-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-utterance-level-representations","title":"Learning utterance-level representations through token-level acoustic latents prediction for Expressive Speech Synthesis","date":"2022-11-01","arxiv_id":"2211.00523","repositories_listed":0,"syntology":null},{"url":null,"slug":"taiwanese-accented-mandarin-and-english-multi","title":"Taiwanese-Accented Mandarin and English Multi-Speaker Talking-Face Synthesis System","date":"2022-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"technology-pipeline-for-large-scale-cross","title":"Technology Pipeline for Large Scale Cross-Lingual Dubbing of Lecture Videos into Multiple Indian Languages","date":"2022-11-01","arxiv_id":"2211.01338","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-importance-of-accurate-alignments-in-end","title":"Towards Developing State-of-the-Art TTS Synthesisers for 13 Indian Languages with Signal Processing aided Alignments","date":"2022-10-31","arxiv_id":"2210.17153","repositories_listed":0,"syntology":null},{"url":null,"slug":"period-vits-variational-inference-with","title":"Period VITS: Variational Inference with Explicit Pitch Modeling for End-to-end Emotional Speech Synthesis","date":"2022-10-28","arxiv_id":"2210.15964","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtuoso-massive-multilingual-speech-text","title":"Virtuoso: Massive Multilingual Speech-Text Joint Semi-Supervised Learning for Text-To-Speech","date":"2022-10-27","arxiv_id":"2210.15447","repositories_listed":0,"syntology":null},{"url":null,"slug":"bloom-library-multimodal-datasets-in-300","title":"Bloom Library: Multimodal Datasets in 300+ Languages for a Variety of Downstream Tasks","date":"2022-10-26","arxiv_id":"2210.14712","repositories_listed":0,"syntology":null},{"url":null,"slug":"redpen-region-and-reason-annotated-dataset-of","title":"RedPen: Region- and Reason-Annotated Dataset of Unnatural Speech","date":"2022-10-26","arxiv_id":"2210.14406","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-based-on-reference","title":"Semi-Supervised Learning Based on Reference Model for Low-resource TTS","date":"2022-10-25","arxiv_id":"2210.14723","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-data-driven-investigation-of-noise-adaptive","title":"A Data-Driven Investigation of Noise-Adaptive Utterance Generation with Linguistic Modification","date":"2022-10-19","arxiv_id":"2210.10252","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-and-effective-unsupervised-speech-1","title":"Simple and Effective Unsupervised Speech Translation","date":"2022-10-18","arxiv_id":"2210.10191","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-speech-synthesizer","title":"Transformer-Based Speech Synthesizer Attribution in an Open Set Scenario","date":"2022-10-14","arxiv_id":"2210.07546","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-affective-speech-synthesis-and","title":"An Overview of Affective Speech Synthesis and Conversion in the Deep Learning Era","date":"2022-10-06","arxiv_id":"2210.03538","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-unsupervised-training-of-few-shot","title":"Fully Unsupervised Training of Few-shot Keyword Spotting","date":"2022-10-06","arxiv_id":"2210.02732","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-multi-scale-expressive-speaking","title":"Unsupervised Multi-scale Expressive Speaking Style Modeling with Hierarchical Context Information for Audiobook Speech Synthesis","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-accented-text-to-speech","title":"Controllable Accented Text-to-Speech Synthesis","date":"2022-09-22","arxiv_id":"2209.10804","repositories_listed":0,"syntology":null},{"url":null,"slug":"epic-tts-models-empirical-pruning","title":"EPIC TTS Models: Empirical Pruning Investigations Characterizing Text-To-Speech Models","date":"2022-09-22","arxiv_id":"2209.10890","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-initial-study-on-birdsong-re-synthesis","title":"An Initial study on Birdsong Re-synthesis Using Neural Vocoders","date":"2022-09-21","arxiv_id":"2209.10479","repositories_listed":0,"syntology":null},{"url":null,"slug":"autolv-automatic-lecture-video-generator","title":"AutoLV: Automatic Lecture Video Generator","date":"2022-09-19","arxiv_id":"2209.08795","repositories_listed":0,"syntology":null},{"url":null,"slug":"decoupled-pronunciation-and-prosody-modeling","title":"Decoupled Pronunciation and Prosody Modeling in Meta-Learning-Based Multilingual Speech Synthesis","date":"2022-09-14","arxiv_id":"2209.06789","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-detection-of-pronunciation-errors","title":"Automated detection of pronunciation errors in non-native English speech employing deep learning","date":"2022-09-13","arxiv_id":"2209.06265","repositories_listed":0,"syntology":null},{"url":null,"slug":"lip-to-speech-synthesis-for-arbitrary","title":"Lip-to-Speech Synthesis for Arbitrary Speakers in the Wild","date":"2022-09-01","arxiv_id":"2209.00642","repositories_listed":0,"syntology":null},{"url":null,"slug":"system-fingerprints-detection-for-deepfake","title":"Audio Deepfake Attribution: An Initial Dataset and Investigation","date":"2022-08-21","arxiv_id":"2208.10489","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-with-mixed-emotions","title":"Speech Synthesis with Mixed Emotions","date":"2022-08-11","arxiv_id":"2208.05890","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-modeling-rising-intonation-in","title":"A Study of Modeling Rising Intonation in Cantonese Neural Speech Synthesis","date":"2022-08-03","arxiv_id":"2208.02189","repositories_listed":0,"syntology":null},{"url":null,"slug":"transplantation-of-conversational-speaking","title":"Transplantation of Conversational Speaking Style with Interjections in Sequence-to-Sequence Speech Synthesis","date":"2022-07-25","arxiv_id":"2207.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-data-generation-by-deep-learning","title":"Controllable Data Generation by Deep Learning: A Review","date":"2022-07-19","arxiv_id":"2207.09542","repositories_listed":0,"syntology":null},{"url":null,"slug":"poetictts-controllable-poetry-reading-for","title":"PoeticTTS -- Controllable Poetry Reading for Literary Studies","date":"2022-07-11","arxiv_id":"2207.05549","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-binaural-speech-synthesis","title":"End-to-End Binaural Speech Synthesis","date":"2022-07-08","arxiv_id":"2207.03697","repositories_listed":0,"syntology":null},{"url":null,"slug":"bert-can-he-predict-contrastive-focus","title":"BERT, can HE predict contrastive focus? Predicting and controlling prominence in neural TTS using a language model","date":"2022-07-04","arxiv_id":"2207.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-and-match-an-empirical-study-on-training","title":"Mix and Match: An Empirical Study on Training Corpus Composition for Polyglot Text-To-Speech (TTS)","date":"2022-07-04","arxiv_id":"2207.01507","repositories_listed":0,"syntology":null},{"url":null,"slug":"computer-assisted-pronunciation-training","title":"Computer-assisted Pronunciation Training -- Speech synthesis is almost all you need","date":"2022-07-02","arxiv_id":"2207.00774","repositories_listed":0,"syntology":null},{"url":null,"slug":"r-melnet-reduced-mel-spectral-modeling-for","title":"R-MelNet: Reduced Mel-Spectral Modeling for Neural TTS","date":"2022-06-30","arxiv_id":"2206.15276","repositories_listed":0,"syntology":null},{"url":null,"slug":"tts-by-tts-2-data-selective-augmentation-for","title":"TTS-by-TTS 2: Data-selective augmentation for neural speech synthesis using ranking support vector machine with variational autoencoder","date":"2022-06-30","arxiv_id":"2206.14984","repositories_listed":0,"syntology":null},{"url":null,"slug":"iemotts-toward-robust-cross-speaker-emotion","title":"iEmoTTS: Toward Robust Cross-Speaker Emotion Transfer and Control for Speech Synthesis based on Disentanglement between Prosody and Timbre","date":"2022-06-29","arxiv_id":"2206.14866","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-variable-and-controllable-duration","title":"Expressive, Variable, and Controllable Duration Modelling in TTS","date":"2022-06-28","arxiv_id":"2206.14165","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-context-aware-style","title":"Self-supervised Context-aware Style Representation for Expressive Speech Synthesis","date":"2022-06-25","arxiv_id":"2206.12559","repositories_listed":0,"syntology":null},{"url":null,"slug":"wolonet-wave-outlooker-for-efficient-and-high","title":"WOLONet: Wave Outlooker for Efficient and High Fidelity Speech Synthesis","date":"2022-06-20","arxiv_id":"2206.09920","repositories_listed":0,"syntology":null},{"url":null,"slug":"acoustic-modeling-for-end-to-end-empathetic","title":"Acoustic Modeling for End-to-End Empathetic Dialogue Speech Synthesis Using Linguistic and Prosodic Contexts of Dialogue History","date":"2022-06-16","arxiv_id":"2206.08039","repositories_listed":0,"syntology":null},{"url":null,"slug":"visagesyntalk-unseen-speaker-video-to-speech","title":"VisageSynTalk: Unseen Speaker Video-to-Speech Synthesis via Speech-Visage Feature Selection","date":"2022-06-15","arxiv_id":"2206.07458","repositories_listed":0,"syntology":null},{"url":null,"slug":"utts-unsupervised-tts-with-conditional","title":"Unsupervised TTS Acoustic Modeling for TTS with Conditional Disentangled Sequential VAE","date":"2022-06-06","arxiv_id":"2206.02512","repositories_listed":0,"syntology":null},{"url":null,"slug":"pronunciation-dictionary-free-multilingual","title":"Pronunciation Dictionary-Free Multilingual Speech Synthesis by Combining Unsupervised and Supervised Phonetic Representations","date":"2022-06-02","arxiv_id":"2206.00951","repositories_listed":0,"syntology":null},{"url":null,"slug":"airo-an-interactive-learning-tool-for","title":"AiRO - an Interactive Learning Tool for Children at Risk of Dyslexia","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bu-tts-an-open-source-bilingual-welsh-english","title":"BU-TTS: An Open-Source, Bilingual Welsh-English, Text-to-Speech Corpus","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"building-open-source-speech-technology-for","title":"Building Open-source Speech Technology for Low-resource Minority Languages with SáMi as an Example – Tools, Methods and Experiments","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-transfer-learning-for-urdu-speech","title":"Exploring Transfer Learning for Urdu Speech Synthesis","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-inter-and-intra-speaker-voice","title":"Investigating Inter- and Intra-speaker Voice Conversion using Audiobooks","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"syntact-a-synthesized-database-of-basic","title":"SyntAct: A Synthesized Database of Basic Emotions","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"macedonian-speech-synthesis-for-assistive","title":"Macedonian Speech Synthesis for Assistive Technology Applications","date":"2022-05-18","arxiv_id":"2205.09198","repositories_listed":0,"syntology":null},{"url":null,"slug":"recab-vae-gumbel-softmax-variational","title":"ReCAB-VAE: Gumbel-Softmax Variational Inference Based on Analytic Divergence","date":"2022-05-09","arxiv_id":"2205.04104","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentive-activation-function-for-improving","title":"Attentive activation function for improving end-to-end spoofing countermeasure systems","date":"2022-05-03","arxiv_id":"2205.01528","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-post-auto-regressive-gan-vocoder-focused-on","title":"A Post Auto-regressive GAN Vocoder Focused on Spectrum Fracture","date":"2022-04-12","arxiv_id":"2204.06086","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-noise-control-for-multispeaker","title":"Fine-grained Noise Control for Multispeaker Speech Synthesis","date":"2022-04-11","arxiv_id":"2204.05070","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-partialspoof-database-and-countermeasures","title":"The PartialSpoof Database and Countermeasures for the Detection of Short Fake Speech Segments Embedded in an Utterance","date":"2022-04-11","arxiv_id":"2204.05177","repositories_listed":0,"syntology":null},{"url":null,"slug":"ddos-a-mos-prediction-framework-utilizing","title":"DDOS: A MOS Prediction Framework utilizing Domain Adaptive Pre-training and Distribution of Opinion Scores","date":"2022-04-07","arxiv_id":"2204.03219","repositories_listed":0,"syntology":null},{"url":null,"slug":"maestro-matched-speech-text-representations","title":"MAESTRO: Matched Speech Text Representations through Modality Matching","date":"2022-04-07","arxiv_id":"2204.03409","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-for-robust-voice","title":"Self-supervised learning for robust voice cloning","date":"2022-04-07","arxiv_id":"2204.03421","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-quantized-prosody-representation","title":"Unsupervised Quantized Prosody Representation for Controllable Speech Synthesis","date":"2022-04-07","arxiv_id":"2204.03238","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-and-effective-unsupervised-speech","title":"Simple and Effective Unsupervised Speech Synthesis","date":"2022-04-06","arxiv_id":"2204.02524","repositories_listed":0,"syntology":null},{"url":"/paper/somos-the-samsung-open-mos-dataset-for-the","slug":"somos-the-samsung-open-mos-dataset-for-the","title":"SOMOS: The Samsung Open MOS Dataset for the Evaluation of Neural Text-to-Speech Synthesis","date":"2022-04-06","arxiv_id":"2204.03040","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-deep-learning-mos-predictors","title":"A Comparison of Deep Learning MOS Predictors for Speech Synthesis Quality","date":"2022-04-05","arxiv_id":"2204.02249","repositories_listed":0,"syntology":null},{"url":null,"slug":"vqtts-high-fidelity-text-to-speech-synthesis","title":"VQTTS: High-Fidelity Text-to-Speech Synthesis with Self-Supervised VQ Acoustic Feature","date":"2022-04-02","arxiv_id":"2204.00768","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaspeech-4-adaptive-text-to-speech-in-zero","title":"AdaSpeech 4: Adaptive Text to Speech in Zero-Shot Scenarios","date":"2022-04-01","arxiv_id":"2204.00436","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-guided-personalized-speech-synthesis","title":"Residual-guided Personalized Speech Synthesis based on Face Image","date":"2022-04-01","arxiv_id":"2204.01672","repositories_listed":0,"syntology":null},{"url":null,"slug":"wavthruvec-latent-speech-representation-as","title":"WavThruVec: Latent speech representation as intermediate features for neural speech synthesis","date":"2022-03-31","arxiv_id":"2203.16930","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-syntax-unicode-x2013-prosody-mapping","title":"Applying Syntax$\\unicode{x2013}$Prosody Mapping Hypothesis and Prosodic Well-Formedness Constraints to Neural Sequence-to-Sequence Speech Synthesis","date":"2022-03-29","arxiv_id":"2203.15276","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-voice-conversion-and-code","title":"Analysis of Voice Conversion and Code-Switching Synthesis Using VQ-VAE","date":"2022-03-28","arxiv_id":"2203.14640","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-expressive-speaking-style-modelling","title":"Towards Expressive Speaking Style Modelling with Hierarchical Context Information for Mandarin Speech Synthesis","date":"2022-03-23","arxiv_id":"2203.12201","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-text-to-speech-pipeline-evaluation","title":"A Text-to-Speech Pipeline, Evaluation Methodology, and Initial Fine-Tuning Results for Child Speech Synthesis","date":"2022-03-22","arxiv_id":"2203.11562","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-speech-recognition-and-synthesis-1","title":"Modeling speech recognition and synthesis simultaneously: Encoding and decoding lexical and sublexical semantic information into speech with no direct access to speech data","date":"2022-03-22","arxiv_id":"2203.11476","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-duration-modeling-for-end-to","title":"AutoTTS: End-to-End Text-to-Speech Synthesis through Differentiable Duration Modeling","date":"2022-03-21","arxiv_id":"2203.11049","repositories_listed":0,"syntology":null},{"url":null,"slug":"adavocoder-adaptive-vocoder-for-custom-voice","title":"AdaVocoder: Adaptive Vocoder for Custom Voice","date":"2022-03-18","arxiv_id":"2203.09825","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-speech-synthesis-perspectives-on","title":"Robotic Speech Synthesis: Perspectives on Interactions, Scenarios, and Ethics","date":"2022-03-17","arxiv_id":"2203.09599","repositories_listed":0,"syntology":null},{"url":null,"slug":"whither-the-priors-for-vocal-interactivity","title":"Whither the Priors for (Vocal) Interactivity?","date":"2022-03-16","arxiv_id":"2203.08578","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-free-non-parallel-many-to-many-voice","title":"Text-free non-parallel many-to-many voice conversion using normalising flows","date":"2022-03-15","arxiv_id":"2203.08009","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-adaption-with-intuitive-prosodic","title":"Speaker Adaption with Intuitive Prosodic Features for Statistical Parametric Speech Synthesis","date":"2022-03-02","arxiv_id":"2203.00951","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-lingual-speech-synthesis-with","title":"Improving Cross-lingual Speech Synthesis with Triplet Training Scheme","date":"2022-02-22","arxiv_id":"2202.10729","repositories_listed":0,"syntology":null},{"url":null,"slug":"vcvts-multi-speaker-video-to-speech-synthesis","title":"VCVTS: Multi-speaker Video-to-Speech synthesis via cross-modal knowledge transfer from voice conversion","date":"2022-02-18","arxiv_id":"2202.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-filter-few-shot-text-to-speech-speaker","title":"Voice Filter: Few-shot text-to-speech speaker adaptation using voice conversion as a post-processing module","date":"2022-02-16","arxiv_id":"2202.08164","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-word-level-prosody-tagging-for","title":"Unsupervised word-level prosody tagging for controllable speech synthesis","date":"2022-02-15","arxiv_id":"2202.07200","repositories_listed":0,"syntology":null},{"url":null,"slug":"partially-fake-audio-detection-by-self","title":"Partially Fake Audio Detection by Self-attention-based Fake Span Discovery","date":"2022-02-14","arxiv_id":"2202.06684","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-performer-score-to-audio-music","title":"Deep Performer: Score-to-Audio Music Performance Synthesis","date":"2022-02-12","arxiv_id":"2202.06034","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-models-of-text","title":"Transformer-based Models of Text Normalization for Speech Applications","date":"2022-02-01","arxiv_id":"2202.00153","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-long-form-voice-cloning-with","title":"Zero-Shot Long-Form Voice Cloning with Dynamic Convolution Attention","date":"2022-01-25","arxiv_id":"2201.10375","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-text-to-speech-using-multi-task","title":"Cross-Lingual Text-to-Speech Using Multi-Task Learning and Speaker Classifier Joint Training","date":"2022-01-20","arxiv_id":"2201.08124","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-speech-synthesis-from-articulatory","title":"Deep Speech Synthesis from Articulatory Features","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ito-taylor-sampling-scheme-for-denoising","title":"Quasi-Taylor Samplers for Diffusion Generative Models based on Ideal Derivatives","date":"2021-12-26","arxiv_id":"2112.13339","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-speaker-multi-style-text-to-speech","title":"Multi-speaker Multi-style Text-to-speech Synthesis With Single-speaker Single-style Training Data Scenarios","date":"2021-12-23","arxiv_id":"2112.12743","repositories_listed":0,"syntology":null},{"url":null,"slug":"zheng-he-yu-zhe-qian-ru-xiang-liang-yu-hou","title":"整合語者嵌入向量與後置濾波器於提升個人化合成語音之語者相似度 (Incorporating Speaker Embedding and Post-Filter Network for Improving Speaker Similarity of Personalized Speech Synthesis System)","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-tts-text-to-speech-with-untranscribed-1","title":"Guided-TTS: A Diffusion Model for Text-to-Speech via Classifier Guidance","date":"2021-11-23","arxiv_id":"2111.11755","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodic-clustering-for-phoneme-level-prosody","title":"Prosodic Clustering for Phoneme-level Prosody Control in End-to-End Speech Synthesis","date":"2021-11-19","arxiv_id":"2111.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-level-style-control-for-expressive-non","title":"Word-Level Style Control for Expressive, Non-attentive Speech Synthesis","date":"2021-11-19","arxiv_id":"2111.10173","repositories_listed":0,"syntology":null}],"record_sha256":"0a86aa5ba9234ca49220430f2c49fb1e056e8c9e0526d05a1e179967cbdfa4bf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}