{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech/papers/10","list_of":"/task/text-to-speech","task":"Text to Speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":15,"rows_per_page":100,"rows":[901,1000],"of":1419,"counts":{"archive_papers_tagged":1419,"with_a_code_link":399,"where_syntology_ran_a_sample":108,"not_listed_spam_title":0,"listed":1419,"listed_where_code_ran":108,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":96,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":96,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech","prev":"/task/text-to-speech/papers/9","next":"/task/text-to-speech/papers/11","papers":[{"url":null,"slug":"pamp-a-unified-framework-boosting-low","title":"MAC: A unified framework boosting low resource automatic speech recognition","date":"2023-02-05","arxiv_id":"2302.03498","repositories_listed":0,"syntology":null},{"url":null,"slug":"uzbektagger-the-rule-based-pos-tagger-for","title":"UzbekTagger: The rule-based POS tagger for Uzbek language","date":"2023-01-30","arxiv_id":"2301.12711","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-granularity-of-prosodic-representations-in","title":"On granularity of prosodic representations in expressive text-to-speech","date":"2023-01-26","arxiv_id":"2301.11446","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-low-resource-accents-without-accent","title":"Modelling low-resource accents without accent-specific TTS frontend","date":"2023-01-11","arxiv_id":"2301.04606","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifyspeech-a-unified-framework-for-zero-shot","title":"UnifySpeech: A Unified Framework for Zero-shot Text-to-Speech and Voice Conversion","date":"2023-01-10","arxiv_id":"2301.03801","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-automated-machine-translation-to","title":"Applying Automated Machine Translation to Educational Video Courses","date":"2023-01-09","arxiv_id":"2301.03141","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-external-off-policy-speech-to-text","title":"Using External Off-Policy Speech-To-Text Mappings in Contextual End-To-End Automated Speech Recognition","date":"2023-01-06","arxiv_id":"2301.02736","repositories_listed":0,"syntology":null},{"url":null,"slug":"revise-self-supervised-speech-resynthesis-1","title":"ReVISE: Self-Supervised Speech Resynthesis With Visual Input for Universal and Generalized Speech Regeneration","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hmm-based-data-augmentation-for-e2e-systems","title":"HMM-based data augmentation for E2E systems for building conversational speech synthesis systems","date":"2022-12-22","arxiv_id":"2212.11982","repositories_listed":0,"syntology":null},{"url":"/paper/revise-self-supervised-speech-resynthesis","slug":"revise-self-supervised-speech-resynthesis","title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","date":"2022-12-21","arxiv_id":"2212.11377","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-quality-of-neural-tts-using","title":"Improving the quality of neural TTS using long-form content and multi-speaker multi-style modeling","date":"2022-12-20","arxiv_id":"2212.10075","repositories_listed":0,"syntology":null},{"url":null,"slug":"tts-guided-training-for-accent-conversion","title":"TTS-Guided Training for Accent Conversion Without Parallel Data","date":"2022-12-20","arxiv_id":"2212.10204","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-japanese-png-bert-language","title":"Investigation of Japanese PnG BERT language model in text-to-speech synthesis for pitch accent language","date":"2022-12-16","arxiv_id":"2212.08321","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-aware-dialog-system-technology","title":"Speech Aware Dialog System Technology Challenge (DSTC11)","date":"2022-12-16","arxiv_id":"2212.08704","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-based-on-latent","title":"Text-to-speech synthesis based on latent variable conversion using diffusion probabilistic model and variational autoencoder","date":"2022-12-16","arxiv_id":"2212.08329","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-deep-speaker-embeddings-for-speaker","title":"Probing Deep Speaker Embeddings for Speaker-related Tasks","date":"2022-12-14","arxiv_id":"2212.07068","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-and-utilization-of-entrainment-on","title":"Analysis and Utilization of Entrainment on Acoustic and Emotion Features in User-agent Dialogue","date":"2022-12-07","arxiv_id":"2212.03398","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-end-to-end-sanskrit-tts-using","title":"Low-Resource End-to-end Sanskrit TTS using Tacotron2, WaveGlow and Transfer Learning","date":"2022-12-07","arxiv_id":"2212.03558","repositories_listed":0,"syntology":null},{"url":null,"slug":"snac-speaker-normalized-affine-coupling-layer","title":"SNAC: Speaker-normalized affine coupling layer in flow-based architecture for zero-shot multi-speaker text-to-speech","date":"2022-11-30","arxiv_id":"2211.16866","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-speech-synthesis-by-learning","title":"Controllable speech synthesis by learning discrete phoneme-level prosodic representations","date":"2022-11-29","arxiv_id":"2211.16307","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-reducing-the-distance-between","title":"Evaluating and reducing the distance between synthetic and real speech distributions","date":"2022-11-29","arxiv_id":"2211.16049","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-expressive-text-to-speech","title":"Contextual Expressive Text-to-Speech","date":"2022-11-26","arxiv_id":"2211.14548","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-incremental-text-to-speech-on-gpus","title":"Efficient Incremental Text-to-Speech on GPUs","date":"2022-11-25","arxiv_id":"2211.13939","repositories_listed":0,"syntology":null},{"url":"/paper/imasc-icfoss-malayalam-speech-corpus","slug":"imasc-icfoss-malayalam-speech-corpus","title":"IMaSC -- ICFOSS Malayalam Speech Corpus","date":"2022-11-23","arxiv_id":"2211.12796","repositories_listed":0,"syntology":null},{"url":null,"slug":"any-speaker-adaptive-text-to-speech-synthesis","title":"Grad-StyleSpeech: Any-speaker Adaptive Text-to-Speech Synthesis with Diffusion Models","date":"2022-11-17","arxiv_id":"2211.09383","repositories_listed":0,"syntology":null},{"url":null,"slug":"back-translation-style-data-augmentation-for-1","title":"Back-Translation-Style Data Augmentation for Mandarin Chinese Polyphone Disambiguation","date":"2022-11-17","arxiv_id":"2211.09495","repositories_listed":0,"syntology":null},{"url":null,"slug":"emodiff-intensity-controllable-emotional-text","title":"EmoDiff: Intensity Controllable Emotional Text-to-Speech with Soft-Label Guidance","date":"2022-11-17","arxiv_id":"2211.09496","repositories_listed":0,"syntology":null},{"url":null,"slug":"sniper-training-variable-sparsity-rate","title":"SNIPER Training: Single-Shot Sparse Training for Text-to-Speech","date":"2022-11-14","arxiv_id":"2211.07283","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-emotional-intensity-controllable","title":"Semi-supervised learning for continuous emotional intensity controllable speech synthesis with disentangled representations","date":"2022-11-11","arxiv_id":"2211.06160","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-on-l2-accents-of-cross","title":"An Empirical Study on L2 Accents of Cross-lingual Text-to-Speech Systems via Vowel Space","date":"2022-11-06","arxiv_id":"2211.03078","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-attention-forcing-for-machine","title":"Parallel Attention Forcing for Machine Translation","date":"2022-11-06","arxiv_id":"2211.03237","repositories_listed":0,"syntology":null},{"url":null,"slug":"stutter-tts-controlled-synthesis-and-improved","title":"Stutter-TTS: Controlled Synthesis and Improved Recognition of Stuttered Speech","date":"2022-11-04","arxiv_id":"2211.09731","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-gender-ambiguous-text-to-speech","title":"Generating Multilingual Gender-Ambiguous Text-to-Speech Voices","date":"2022-11-01","arxiv_id":"2211.00375","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-content-aware-neural-text-to","title":"Investigating Content-Aware Neural Text-To-Speech MOS Prediction Using Prosodic and Linguistic Features","date":"2022-11-01","arxiv_id":"2211.00342","repositories_listed":0,"syntology":null},{"url":null,"slug":"technology-pipeline-for-large-scale-cross","title":"Technology Pipeline for Large Scale Cross-Lingual Dubbing of Lecture Videos into Multiple Indian Languages","date":"2022-11-01","arxiv_id":"2211.01338","repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-automatic-speaker-verification-and","title":"Combining Automatic Speaker Verification and Prosody Analysis for Synthetic Speech Detection","date":"2022-10-31","arxiv_id":"2210.17222","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-text-to-speech-with-flow-based","title":"Cross-lingual Text-To-Speech with Flow-based Voice Conversion for Improved Pronunciation","date":"2022-10-31","arxiv_id":"2210.17264","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-state-space-decoder-for-speech","title":"Structured State Space Decoder for Speech Recognition and Synthesis","date":"2022-10-31","arxiv_id":"2210.17098","repositories_listed":0,"syntology":null},{"url":null,"slug":"period-vits-variational-inference-with","title":"Period VITS: Variational Inference with Explicit Pitch Modeling for End-to-end Emotional Speech Synthesis","date":"2022-10-28","arxiv_id":"2210.15964","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-adapters-for-few-shot-text-to-speech","title":"Residual Adapters for Few-Shot Text-to-Speech Speaker Adaptation","date":"2022-10-28","arxiv_id":"2210.15868","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-text-based-voice-editing","title":"Towards zero-shot Text-based voice editing using acoustic context conditioning, utterance embeddings, and reference encoders","date":"2022-10-28","arxiv_id":"2210.16045","repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-intensity-control-for-accented-text","title":"Explicit Intensity Control for Accented Text-to-speech","date":"2022-10-27","arxiv_id":"2210.15364","repositories_listed":0,"syntology":null},{"url":null,"slug":"virtuoso-massive-multilingual-speech-text","title":"Virtuoso: Massive Multilingual Speech-Text Joint Semi-Supervised Learning for Text-To-Speech","date":"2022-10-27","arxiv_id":"2210.15447","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-to-speech-translation","title":"Improving Speech-to-Speech Translation Through Unlabeled Text","date":"2022-10-26","arxiv_id":"2210.14514","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapitch-adaption-multi-speaker-text-to","title":"Adapitch: Adaption Multi-Speaker Text-to-Speech Conditioned on Pitch Disentangling with Untranscribed Data","date":"2022-10-25","arxiv_id":"2210.13803","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-based-on-reference","title":"Semi-Supervised Learning Based on Reference Model for Low-resource TTS","date":"2022-10-25","arxiv_id":"2210.14723","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiently-trained-mongolian-text-to-speech","title":"Efficiently Trained Low-Resource Mongolian Text-to-Speech System Based On FullConv-TTS","date":"2022-10-24","arxiv_id":"2211.01948","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-re-calibration-of-channel-wise","title":"Adaptive re-calibration of channel-wise features for Adversarial Audio Classification","date":"2022-10-21","arxiv_id":"2210.11722","repositories_listed":0,"syntology":null},{"url":null,"slug":"levoice-asr-systems-for-the-iscslp-2022","title":"LeVoice ASR Systems for the ISCSLP 2022 Intelligent Cockpit Speech Recognition Challenge","date":"2022-10-14","arxiv_id":"2210.07749","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-avatar-an-automatic-presentation","title":"Pre-Avatar: An Automatic Presentation Generation Framework Leveraging Talking Avatar","date":"2022-10-13","arxiv_id":"2210.06877","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-speaker-consistency-learning","title":"Adversarial Speaker-Consistency Learning Using Untranscribed Speech Data for Zero-Shot Multi-Speaker Text-to-Speech","date":"2022-10-12","arxiv_id":"2210.05979","repositories_listed":0,"syntology":null},{"url":null,"slug":"squid-measuring-speech-naturalness-in-many","title":"SQuId: Measuring Speech Naturalness in Many Languages","date":"2022-10-12","arxiv_id":"2210.06324","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-affective-speech-synthesis-and","title":"An Overview of Affective Speech Synthesis and Conversion in the Deep Learning Era","date":"2022-10-06","arxiv_id":"2210.03538","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-multi-scale-expressive-speaking","title":"Unsupervised Multi-scale Expressive Speaking Style Modeling with Hierarchical Context Information for Audiobook Speech Synthesis","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-adversarial-training-algorithm-for","title":"Multi-Task Adversarial Training Algorithm for Multi-Speaker Neural Text-to-Speech","date":"2022-09-26","arxiv_id":"2209.12549","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-accented-text-to-speech","title":"Controllable Accented Text-to-Speech Synthesis","date":"2022-09-22","arxiv_id":"2209.10804","repositories_listed":0,"syntology":null},{"url":null,"slug":"epic-tts-models-empirical-pruning","title":"EPIC TTS Models: Empirical Pruning Investigations Characterizing Text-To-Speech Models","date":"2022-09-22","arxiv_id":"2209.10890","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-rater-and-system-metadata-to-explain","title":"Using Rater and System Metadata to Explain Variance in the VoiceMOS Challenge 2022 Dataset","date":"2022-09-14","arxiv_id":"2209.06358","repositories_listed":0,"syntology":null},{"url":null,"slug":"sanip-shopping-assistant-and-navigation-for","title":"SANIP: Shopping Assistant and Navigation for the visually impaired","date":"2022-09-08","arxiv_id":"2209.03570","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-standard-vietnamese-word-detection-and","title":"Non-Standard Vietnamese Word Detection and Normalization for Text-to-Speech","date":"2022-09-07","arxiv_id":"2209.02971","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-contextual-recognition-of-rare","title":"Improving Contextual Recognition of Rare Words with an Alternate Spelling Prediction Model","date":"2022-09-02","arxiv_id":"2209.01250","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-moocs-for-lip-reading-using-synthetic","title":"Towards MOOCs for Lipreading: Using Synthetic Talking Heads to Train Humans in Lipreading at Scale","date":"2022-08-21","arxiv_id":"2208.09796","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-with-mixed-emotions","title":"Speech Synthesis with Mixed Emotions","date":"2022-08-11","arxiv_id":"2208.05890","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-modeling-rising-intonation-in","title":"A Study of Modeling Rising Intonation in Cantonese Neural Speech Synthesis","date":"2022-08-03","arxiv_id":"2208.02189","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-data-no-problem-low-resource-language","title":"Low-data? No problem: low-resource, language-agnostic conversational text-to-speech via F0-conditioned data augmentation","date":"2022-07-29","arxiv_id":"2207.14607","repositories_listed":0,"syntology":null},{"url":null,"slug":"transplantation-of-conversational-speaking","title":"Transplantation of Conversational Speaking Style with Interjections in Sequence-to-Sequence Speech Synthesis","date":"2022-07-25","arxiv_id":"2207.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cyclical-approach-to-synthetic-and-natural","title":"A Cyclical Approach to Synthetic and Natural Speech Mismatch Refinement of Neural Post-filter for Low-cost Text-to-speech System","date":"2022-07-13","arxiv_id":"2207.05913","repositories_listed":0,"syntology":null},{"url":null,"slug":"satts-speaker-attractor-text-to-speech","title":"SATTS: Speaker Attractor Text to Speech, Learning to Speak by Learning to Separate","date":"2022-07-13","arxiv_id":"2207.06011","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-driven-emotional-style-control-and-cross","title":"Text-driven Emotional Style Control and Cross-speaker Style Transfer in Neural TTS","date":"2022-07-13","arxiv_id":"2207.06000","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-recognition-modeling-from","title":"End-to-end speech recognition modeling from de-identified data","date":"2022-07-12","arxiv_id":"2207.05469","repositories_listed":0,"syntology":null},{"url":null,"slug":"huqariq-a-multilingual-speech-corpus-of","title":"Huqariq: A Multilingual Speech Corpus of Native Languages of Peru for Speech Recognition","date":"2022-07-12","arxiv_id":"2207.05498","repositories_listed":0,"syntology":null},{"url":null,"slug":"lip-lightweight-intelligent-preprocessor-for","title":"LIP: Lightweight Intelligent Preprocessor for meaningful text-to-speech","date":"2022-07-11","arxiv_id":"2207.07118","repositories_listed":0,"syntology":null},{"url":null,"slug":"bert-can-he-predict-contrastive-focus","title":"BERT, can HE predict contrastive focus? Predicting and controlling prominence in neural TTS using a language model","date":"2022-07-04","arxiv_id":"2207.01718","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-and-match-an-empirical-study-on-training","title":"Mix and Match: An Empirical Study on Training Corpus Composition for Polyglot Text-To-Speech (TTS)","date":"2022-07-04","arxiv_id":"2207.01507","repositories_listed":0,"syntology":null},{"url":null,"slug":"unify-and-conquer-how-phonetic-feature","title":"Unify and Conquer: How Phonetic Feature Representation Affects Polyglot Text-To-Speech (TTS)","date":"2022-07-04","arxiv_id":"2207.01547","repositories_listed":0,"syntology":null},{"url":null,"slug":"computer-assisted-pronunciation-training","title":"Computer-assisted Pronunciation Training -- Speech synthesis is almost all you need","date":"2022-07-02","arxiv_id":"2207.00774","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-polyphone-bert-for-polyphone-disambiguation","title":"A Polyphone BERT for Polyphone Disambiguation in Mandarin Chinese","date":"2022-07-01","arxiv_id":"2207.12089","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-evaluation-of-speaker-similarity","title":"Automatic Evaluation of Speaker Similarity","date":"2022-07-01","arxiv_id":"2207.00344","repositories_listed":0,"syntology":null},{"url":null,"slug":"empathic-machines-using-intermediate-features-1","title":"Empathic Machines: Using Intermediate Features as Levers to Emulate Emotions in Text-To-Speech Systems","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-bilingual-grapheme-to-phoneme-conversion","title":"Fast Bilingual Grapheme-To-Phoneme Conversion","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"r-melnet-reduced-mel-spectral-modeling-for","title":"R-MelNet: Reduced Mel-Spectral Modeling for Neural TTS","date":"2022-06-30","arxiv_id":"2206.15276","repositories_listed":0,"syntology":null},{"url":null,"slug":"tts-by-tts-2-data-selective-augmentation-for","title":"TTS-by-TTS 2: Data-selective augmentation for neural speech synthesis using ranking support vector machine with variational autoencoder","date":"2022-06-30","arxiv_id":"2206.14984","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-deliberation-by-text-only-and-semi","title":"Improving Deliberation by Text-Only and Semi-Supervised Training","date":"2022-06-29","arxiv_id":"2206.14716","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-and-effective-multi-sentence-tts-with","title":"Simple and Effective Multi-sentence TTS with Expressive and Coherent Prosody","date":"2022-06-29","arxiv_id":"2206.14643","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-speech-representations-for-the","title":"Comparison of Speech Representations for the MOS Prediction System","date":"2022-06-28","arxiv_id":"2206.13817","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-variable-and-controllable-duration","title":"Expressive, Variable, and Controllable Duration Modelling in TTS","date":"2022-06-28","arxiv_id":"2206.14165","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-cross-lingual-tts-using-transferable","title":"Few-Shot Cross-Lingual TTS Using Transferable Phoneme Embedding","date":"2022-06-27","arxiv_id":"2206.15427","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-personalized-non-speech","title":"Synthesizing Personalized Non-speech Vocalization from Discrete Speech Representations","date":"2022-06-25","arxiv_id":"2206.12662","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-text-to-speech-based-on-latent","title":"End-to-End Text-to-Speech Based on Latent Representation of Speaking Styles Using Spontaneous Dialogue","date":"2022-06-24","arxiv_id":"2206.12040","repositories_listed":0,"syntology":null},{"url":null,"slug":"sane-tts-stable-and-natural-end-to-end","title":"SANE-TTS: Stable And Natural End-to-End Multilingual Text-to-Speech","date":"2022-06-24","arxiv_id":"2206.12132","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-baseline-for-domain-adaptation-in-1","title":"A Simple Baseline for Domain Adaptation in End to End ASR Systems Using Synthetic Data","date":"2022-06-22","arxiv_id":"2206.13240","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-speaker-adaptation-for-dnn","title":"Human-in-the-loop Speaker Adaptation for DNN-based Multi-speaker TTS","date":"2022-06-21","arxiv_id":"2206.10256","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-optimizing-ocr-for-accessibility","title":"Towards Optimizing OCR for Accessibility","date":"2022-06-21","arxiv_id":"2206.10254","repositories_listed":0,"syntology":null},{"url":null,"slug":"natiq-an-end-to-end-text-to-speech-system-for","title":"NatiQ: An End-to-end Text-to-Speech System for Arabic","date":"2022-06-15","arxiv_id":"2206.07373","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-chinese-dialect-tts-frontend-with-non","title":"A Novel Chinese Dialect TTS Frontend with Non-Autoregressive Neural Machine Translation","date":"2022-06-10","arxiv_id":"2206.04922","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-dubbing-lip-synchronous-voice-preserving","title":"Face-Dubbing++: Lip-Synchronous, Voice Preserving Translation of Videos","date":"2022-06-09","arxiv_id":"2206.04523","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexlip-a-controllable-text-to-lip-system","title":"FlexLip: A Controllable Text-to-Lip System","date":"2022-06-07","arxiv_id":"2206.03206","repositories_listed":0,"syntology":null},{"url":null,"slug":"utts-unsupervised-tts-with-conditional","title":"Unsupervised TTS Acoustic Modeling for TTS with Conditional Disentangled Sequential VAE","date":"2022-06-06","arxiv_id":"2206.02512","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiobook-dialogues-as-training-data-for","title":"Audiobook Dialogues as Training Data for Conversational Style Synthetic Voices","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"bu-tts-an-open-source-bilingual-welsh-english","title":"BU-TTS: An Open-Source, Bilingual Welsh-English, Text-to-Speech Corpus","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"8aa9bbb297f5ccbbac67e8ae9d01faea03a5a57ddb3549c8421859f963ef5880","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}