{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-1/papers/9","list_of":"/task/text-to-speech-1","task":"text-to-speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":15,"rows_per_page":100,"rows":[801,900],"of":1413,"counts":{"archive_papers_tagged":1413,"with_a_code_link":395,"where_syntology_ran_a_sample":106,"not_listed_spam_title":0,"listed":1413,"listed_where_code_ran":106,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-1","prev":"/task/text-to-speech-1/papers/8","next":"/task/text-to-speech-1/papers/10","papers":[{"url":null,"slug":"mulantts-the-microsoft-speech-synthesis","title":"MuLanTTS: The Microsoft Speech Synthesis System for Blizzard Challenge 2023","date":"2023-09-06","arxiv_id":"2309.02743","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompttts-2-describing-and-generating-voices","title":"PromptTTS 2: Describing and Generating Voices with Text Prompt","date":"2023-09-05","arxiv_id":"2309.02285","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-pretrained-language","title":"A Comparative Analysis of Pretrained Language Models for Text-to-Speech","date":"2023-09-04","arxiv_id":"2309.01576","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpsp-learning-speech-concepts-from-phoneme","title":"Learning Speech Representation From Contrastive Token-Acoustic Pretraining","date":"2023-09-01","arxiv_id":"2309.00424","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-fruitshell-french-synthesis-system-at-the","title":"The FruitShell French synthesis system at the Blizzard 2023 Challenge","date":"2023-09-01","arxiv_id":"2309.00223","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-mandarin-prosodic-structure","title":"Improving Mandarin Prosodic Structure Prediction with Multi-level Contextual Information","date":"2023-08-31","arxiv_id":"2308.16577","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spontaneous-style-modeling-with-semi","title":"Towards Spontaneous Style Modeling with Semi-supervised Pre-training for Conversational Text-to-Speech Synthesis","date":"2023-08-31","arxiv_id":"2308.16593","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-deepzen-speech-synthesis-system-for","title":"The DeepZen Speech Synthesis System for Blizzard Challenge 2023","date":"2023-08-30","arxiv_id":"2308.15945","repositories_listed":0,"syntology":null},{"url":null,"slug":"pruning-self-attention-for-zero-shot-multi","title":"Pruning Self-Attention for Zero-Shot Multi-Speaker Text-to-Speech","date":"2023-08-28","arxiv_id":"2308.14909","repositories_listed":0,"syntology":null},{"url":null,"slug":"rep2wav-noise-robust-text-to-speech-using","title":"Rep2wav: Noise Robust text-to-speech Using self-supervised representations","date":"2023-08-28","arxiv_id":"2308.14553","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-zero-shot-speaker-adaptive","title":"Generalizable Zero-Shot Speaker Adaptive Speech Synthesis with Disentangled Representations","date":"2023-08-24","arxiv_id":"2308.13007","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-gradspeech-towards-diffusion-based","title":"Multi-GradSpeech: Towards Diffusion-based Multi-Speaker Text-to-speech Using Consistent Diffusion Models","date":"2023-08-21","arxiv_id":"2308.10428","repositories_listed":0,"syntology":null},{"url":null,"slug":"affectecho-speaker-independent-and-language","title":"AffectEcho: Speaker Independent and Language-Agnostic Emotion and Affect Transfer for Speech Synthesis","date":"2023-08-16","arxiv_id":"2308.08577","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechx-neural-codec-language-model-as-a","title":"SpeechX: Neural Codec Language Model as a Versatile Speech Transformer","date":"2023-08-14","arxiv_id":"2308.06873","repositories_listed":0,"syntology":null},{"url":null,"slug":"saltts-leveraging-self-supervised-speech","title":"SALTTS: Leveraging Self-Supervised Speech Representations for improved Text-to-Speech Synthesis","date":"2023-08-02","arxiv_id":"2308.01018","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-normalizing-flows-and-diffusion","title":"Comparing normalizing flows and diffusion models for prosody and acoustic modelling in text-to-speech","date":"2023-07-31","arxiv_id":"2307.16679","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-grapheme-to-phoneme-conversion-by","title":"Improving grapheme-to-phoneme conversion by learning pronunciations from speech recordings","date":"2023-07-31","arxiv_id":"2307.16643","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-context-based-pronunciation","title":"Multilingual context-based pronunciation learning for Text-to-Speech","date":"2023-07-31","arxiv_id":"2307.16709","repositories_listed":0,"syntology":null},{"url":null,"slug":"metts-multilingual-emotional-text-to-speech","title":"METTS: Multilingual Emotional Text-to-Speech by Cross-speaker and Cross-lingual Emotion Transfer","date":"2023-07-29","arxiv_id":"2307.15951","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimally-supervised-speech-synthesis-with","title":"Minimally-Supervised Speech Synthesis with Conditional Diffusion Model and Language Model: A Comparative Study of Semantic Coding","date":"2023-07-28","arxiv_id":"2307.15484","repositories_listed":0,"syntology":null},{"url":null,"slug":"slmgan-exploiting-speech-language-model","title":"SLMGAN: Exploiting Speech Language Model Representations for Unsupervised Zero-Shot Voice Conversion in GANs","date":"2023-07-18","arxiv_id":"2307.09435","repositories_listed":0,"syntology":null},{"url":null,"slug":"mega-tts-2-zero-shot-text-to-speech-with","title":"Mega-TTS 2: Boosting Prompting Mechanisms for Zero-Shot Speech Synthesis","date":"2023-07-14","arxiv_id":"2307.07218","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-emphasis-with-zero-data-for-text","title":"Controllable Emphasis with zero data for text-to-speech","date":"2023-07-13","arxiv_id":"2307.07062","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-use-of-self-supervised-speech","title":"On the Use of Self-Supervised Speech Representations in Spontaneous Speech Synthesis","date":"2023-07-11","arxiv_id":"2307.05132","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-eye-for-the-blind","title":"Artificial Eye for the Blind","date":"2023-07-07","arxiv_id":"2308.00801","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextspeech-expressive-and-efficient-text","title":"ContextSpeech: Expressive and Efficient Text-to-Speech for Paragraph Reading","date":"2023-07-03","arxiv_id":"2307.00782","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-quality-automatic-voice-over-with","title":"High-Quality Automatic Voice Over with Accurate Alignment: Supervision through Self-Supervised Discrete Speech Units","date":"2023-06-29","arxiv_id":"2306.17005","repositories_listed":0,"syntology":null},{"url":null,"slug":"genertts-pronunciation-disentanglement-for","title":"GenerTTS: Pronunciation Disentanglement for Timbre and Style Generalization in Cross-Lingual Text-to-Speech","date":"2023-06-27","arxiv_id":"2306.15304","repositories_listed":0,"syntology":null},{"url":null,"slug":"dse-tts-dual-speaker-embedding-for-cross","title":"DSE-TTS: Dual Speaker Embedding for Cross-Lingual Text-to-Speech","date":"2023-06-25","arxiv_id":"2306.14145","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-aware-text-to-speech","title":"Visual-Aware Text-to-Speech","date":"2023-06-21","arxiv_id":"2306.12020","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-machine-dubbing-through-phrase","title":"Expressive Machine Dubbing Through Phrase-level Cross-lingual Prosody Transfer","date":"2023-06-20","arxiv_id":"2306.11662","repositories_listed":0,"syntology":null},{"url":null,"slug":"cml-tts-a-multilingual-dataset-for-speech","title":"CML-TTS A Multilingual Dataset for Speech Synthesis in Low-Resource Languages","date":"2023-06-16","arxiv_id":"2306.10097","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-text-to-speech-using-specific","title":"Low-Resource Text-to-Speech Using Specific Data and Noise Augmentation","date":"2023-06-16","arxiv_id":"2306.10152","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-code-switching-and-named-entity","title":"Improving Code-Switching and Named Entity Recognition in ASR with Speech Editing based Data Augmentation","date":"2023-06-14","arxiv_id":"2306.08588","repositories_listed":0,"syntology":null},{"url":null,"slug":"pausespeech-natural-speech-synthesis-via-pre","title":"PauseSpeech: Natural Speech Synthesis via Pre-trained Language Model and Pause-based Prosody Modeling","date":"2023-06-13","arxiv_id":"2306.07489","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-emotional-representations-from","title":"Learning Emotional Representations from Imbalanced Speech Data for Speech Emotion Recognition and Emotional Text-to-Speech","date":"2023-06-09","arxiv_id":"2306.05709","repositories_listed":0,"syntology":null},{"url":null,"slug":"ada-tta-towards-adaptive-high-quality-text-to","title":"Ada-TTA: Towards Adaptive High-Quality Text-to-Talking Avatar Synthesis","date":"2023-06-06","arxiv_id":"2306.03504","repositories_listed":0,"syntology":null},{"url":null,"slug":"mega-tts-zero-shot-text-to-speech-at-scale","title":"Mega-TTS: Zero-Shot Text-to-Speech at Scale with Intrinsic Inductive Bias","date":"2023-06-06","arxiv_id":"2306.03509","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-transfer-learning-for-phrase","title":"Cross-Lingual Transfer Learning for Phrase Break Prediction with Multilingual Language Model","date":"2023-06-05","arxiv_id":"2306.02579","repositories_listed":0,"syntology":null},{"url":null,"slug":"rhythm-controllable-attention-with-high","title":"Rhythm-controllable Attention with High Robustness for Long Sentence Speech Synthesis","date":"2023-06-05","arxiv_id":"2306.02593","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-effects-of-input-type-and-pronunciation","title":"The Effects of Input Type and Pronunciation Dictionary Usage in Transfer Learning for Low-Resource Text-to-Speech","date":"2023-06-01","arxiv_id":"2306.00535","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-pipeline-for-swiss-german-a","title":"Text-to-Speech Pipeline for Swiss German -- A comparison","date":"2023-05-31","arxiv_id":"2305.19750","repositories_listed":0,"syntology":null},{"url":null,"slug":"libritts-r-a-restored-multi-speaker-text-to","title":"LibriTTS-R: A Restored Multi-Speaker Text-to-Speech Corpus","date":"2023-05-30","arxiv_id":"2305.18802","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-a-voice-unified-voice-synthesis-with","title":"Make-A-Voice: Unified Voice Synthesis With Discrete Representation","date":"2023-05-30","arxiv_id":"2305.19269","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-fine-tuning-strategies-for","title":"Resource-Efficient Fine-Tuning Strategies for Automatic MOS Prediction in Text-to-Speech for Low-Resource Languages","date":"2023-05-30","arxiv_id":"2305.19396","repositories_listed":0,"syntology":null},{"url":null,"slug":"stt4sg-350-a-speech-corpus-for-all-swiss","title":"STT4SG-350: A Speech Corpus for All Swiss German Dialect Regions","date":"2023-05-30","arxiv_id":"2305.18855","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-selection-of-text-to-speech-data-to","title":"Towards Selection of Text-to-speech Data to Augment ASR Training","date":"2023-05-30","arxiv_id":"2306.00998","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-evaluation-of-turn-taking-cues-in","title":"Automatic Evaluation of Turn-taking Cues in Conversational Speech Synthesis","date":"2023-05-29","arxiv_id":"2305.17971","repositories_listed":0,"syntology":null},{"url":null,"slug":"disfluencyfixer-a-tool-to-enhance-language","title":"DisfluencyFixer: A tool to enhance Language Learning through Speech To Speech Disfluency Correction","date":"2023-05-26","arxiv_id":"2305.16957","repositories_listed":0,"syntology":null},{"url":null,"slug":"viola-unified-codec-language-models-for","title":"VioLA: Unified Codec Language Models for Speech Recognition, Synthesis, and Translation","date":"2023-05-25","arxiv_id":"2305.16107","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-arabic-ai-with-large-language","title":"LAraBench: Benchmarking Arabic AI with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14982","repositories_listed":0,"syntology":null},{"url":null,"slug":"zet-speech-zero-shot-adaptive-emotion","title":"ZET-Speech: Zero-shot adaptive Emotion-controllable Text-to-Speech Synthesis with Diffusion and Style-based Models","date":"2023-05-23","arxiv_id":"2305.13831","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-generation-with-speech-synthesis-for-asr","title":"Text Generation with Speech Synthesis for ASR Data Augmentation","date":"2023-05-22","arxiv_id":"2305.16333","repositories_listed":0,"syntology":null},{"url":null,"slug":"vit-tts-visual-text-to-speech-with-scalable","title":"ViT-TTS: Visual Text-to-Speech with Scalable Diffusion Transformer","date":"2023-05-22","arxiv_id":"2305.12708","repositories_listed":0,"syntology":null},{"url":null,"slug":"vakta-setu-a-speech-to-speech-machine","title":"VAKTA-SETU: A Speech-to-Speech Machine Translation Service in Select Indic Languages","date":"2023-05-21","arxiv_id":"2305.12518","repositories_listed":0,"syntology":null},{"url":null,"slug":"comedicspeech-text-to-speech-for-stand-up","title":"ComedicSpeech: Text To Speech For Stand-up Comedies in Low-Resource Scenarios","date":"2023-05-20","arxiv_id":"2305.12200","repositories_listed":0,"syntology":null},{"url":null,"slug":"mparrottts-multilingual-multi-speaker-text-to","title":"MParrotTTS: Multilingual Multi-speaker Text to Speech Synthesis in Low Resource Setting","date":"2023-05-19","arxiv_id":"2305.11926","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-front-end-framework-for-english","title":"A unified front-end framework for English text-to-speech synthesis","date":"2023-05-18","arxiv_id":"2305.10666","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-redaction-from-conditional-generative","title":"Data Redaction from Conditional Generative Models","date":"2023-05-18","arxiv_id":"2305.11351","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastfit-towards-real-time-iterative-neural","title":"FastFit: Towards Real-Time Iterative Neural Vocoder by Replacing U-Net Encoder With Multiple STFTs","date":"2023-05-18","arxiv_id":"2305.10823","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-a-large-language-model-to-control","title":"Controllable Speaking Styles Using a Large Language Model","date":"2023-05-17","arxiv_id":"2305.10321","repositories_listed":0,"syntology":null},{"url":null,"slug":"accented-text-to-speech-synthesis-with","title":"Accented Text-to-Speech Synthesis with Limited Data","date":"2023-05-08","arxiv_id":"2305.04816","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2-ctts-end-to-end-multi-scale-multi-modal","title":"M2-CTTS: End-to-End Multi-scale Multi-modal Conversational Text-to-Speech Synthesis","date":"2023-05-03","arxiv_id":"2305.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-learning-techniques-for-3","title":"A Review of Deep Learning Techniques for Speech Processing","date":"2023-04-30","arxiv_id":"2305.00359","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-synthesis","title":"Zero-shot text-to-speech synthesis conditioned using self-supervised speech representation model","date":"2023-04-24","arxiv_id":"2304.11976","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffvoice-text-to-speech-with-latent","title":"DiffVoice: Text-to-Speech with Latent Diffusion","date":"2023-04-23","arxiv_id":"2304.11750","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-virtual-simulation-pilot-agent-for-training","title":"A Virtual Simulation-Pilot Agent for Training of Air Traffic Controllers","date":"2023-04-16","arxiv_id":"2304.07842","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-to-speech-translation-with","title":"Enhancing Speech-to-Speech Translation with Multiple TTS Targets","date":"2023-04-10","arxiv_id":"2304.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"armantts-single-speaker-persian-dataset","title":"ArmanTTS single-speaker Persian dataset","date":"2023-04-07","arxiv_id":"2304.03585","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-prosody-prediction-for-expressive","title":"Ensemble prosody prediction for expressive speech synthesis","date":"2023-04-03","arxiv_id":"2304.00714","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-is-all-you-need-personalizing-asr-models","title":"Text is All You Need: Personalizing ASR Models using Controllable Speech Synthesis","date":"2023-03-27","arxiv_id":"2303.14885","repositories_listed":0,"syntology":null},{"url":null,"slug":"wave-u-net-discriminator-fast-and-lightweight","title":"Wave-U-Net Discriminator: Fast and Lightweight Discriminator for Generative Adversarial Network-Based Speech Synthesis","date":"2023-03-24","arxiv_id":"2303.13909","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-diffusion-model-for-speech-synthesis-a","title":"A Survey on Audio Diffusion Models: Text To Speech Synthesis and Enhancement in Generative AI","date":"2023-03-23","arxiv_id":"2303.13336","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-switching-text-generation-and-injection","title":"Code-Switching Text Generation and Injection in Mandarin-English ASR","date":"2023-03-20","arxiv_id":"2303.10949","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-speaker-emotion-transfer-by","title":"Cross-speaker Emotion Transfer by Manipulating Speech Style Latents","date":"2023-03-15","arxiv_id":"2303.08329","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-high-dimensional-data-with-sparse","title":"Controllable Prosody Generation With Partial Inputs","date":"2023-03-14","arxiv_id":"2303.09446","repositories_listed":0,"syntology":null},{"url":null,"slug":"qi-tts-questioning-intonation-control-for","title":"QI-TTS: Questioning Intonation Control for Emotional Speech Synthesis","date":"2023-03-14","arxiv_id":"2303.07682","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-end-to-end-neural-network-for-image-to","title":"An End-to-End Neural Network for Image-to-Audio Transformation","date":"2023-03-10","arxiv_id":"2303.06078","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-ecg-12-lead-electrocardiogram","title":"Text-to-ECG: 12-Lead Electrocardiogram Synthesis conditioned on Clinical Text Reports","date":"2023-03-09","arxiv_id":"2303.09395","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-prosody-transfer-models-transfer-prosody","title":"Do Prosody Transfer Models Transfer Prosody?","date":"2023-03-07","arxiv_id":"2303.04289","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundationtts-text-to-speech-for-asr","title":"FoundationTTS: Text-to-Speech for ASR Customization with Generative Language Model","date":"2023-03-06","arxiv_id":"2303.02939","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-emotional-control-of-text-to","title":"Fine-grained Emotional Control of Text-To-Speech: Learning To Rank Inter- And Intra-Class Emotion Intensities","date":"2023-03-02","arxiv_id":"2303.01508","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-text-corpora-for-end-to-end","title":"Leveraging Large Text Corpora for End-to-End Speech Summarization","date":"2023-03-02","arxiv_id":"2303.00978","repositories_listed":0,"syntology":null},{"url":null,"slug":"liteg2p-a-fast-light-and-high-accuracy-model","title":"LiteG2P: A fast, light and high accuracy model for grapheme-to-phoneme conversion","date":"2023-03-02","arxiv_id":"2303.01086","repositories_listed":0,"syntology":null},{"url":null,"slug":"dtw-siamesenet-dynamic-time-warped-siamese","title":"DTW-SiameseNet: Dynamic Time Warped Siamese Network for Mispronunciation Detection and Correction","date":"2023-03-01","arxiv_id":"2303.00171","repositories_listed":0,"syntology":null},{"url":null,"slug":"parrottts-text-to-speech-synthesis-by","title":"ParrotTTS: Text-to-Speech synthesis by exploiting self-supervised representations","date":"2023-03-01","arxiv_id":"2303.01261","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-heteronym-resolution-pipeline-using","title":"Automatic Heteronym Resolution Pipeline Using RAD-TTS Aligners","date":"2023-02-28","arxiv_id":"2302.14523","repositories_listed":0,"syntology":null},{"url":null,"slug":"clartts-an-open-source-classical-arabic-text","title":"ClArTTS: An Open-Source Classical Arabic Text-to-Speech Corpus","date":"2023-02-28","arxiv_id":"2303.00069","repositories_listed":0,"syntology":null},{"url":null,"slug":"crossspeech-speaker-independent-acoustic","title":"CrossSpeech: Speaker-independent Acoustic Representation for Cross-lingual Speech Synthesis","date":"2023-02-28","arxiv_id":"2302.14370","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniflg-unified-facial-landmark-generator-from","title":"UniFLG: Unified Facial Landmark Generator from Text or Speech","date":"2023-02-28","arxiv_id":"2302.14337","repositories_listed":0,"syntology":null},{"url":null,"slug":"duration-aware-pause-insertion-using-pre","title":"Duration-aware pause insertion using pre-trained language model for multi-speaker text-to-speech","date":"2023-02-27","arxiv_id":"2302.13652","repositories_listed":0,"syntology":null},{"url":null,"slug":"varianceflow-high-quality-and-controllable","title":"Varianceflow: High-Quality and Controllable Text-to-Speech using Variance Information via Normalizing Flow","date":"2023-02-27","arxiv_id":"2302.13458","repositories_listed":0,"syntology":null},{"url":null,"slug":"emphasizing-unseen-words-new-vocabulary","title":"Emphasizing Unseen Words: New Vocabulary Acquisition for End-to-End Speech Recognition","date":"2023-02-20","arxiv_id":"2302.09723","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-small-footprint-hybrid-hmm-hifigan","title":"Fast and small footprint Hybrid HMM-HiFiGAN based system for speech synthesis in Indian languages","date":"2023-02-13","arxiv_id":"2302.06227","repositories_listed":0,"syntology":null},{"url":null,"slug":"pamp-a-unified-framework-boosting-low","title":"MAC: A unified framework boosting low resource automatic speech recognition","date":"2023-02-05","arxiv_id":"2302.03498","repositories_listed":0,"syntology":null},{"url":null,"slug":"uzbektagger-the-rule-based-pos-tagger-for","title":"UzbekTagger: The rule-based POS tagger for Uzbek language","date":"2023-01-30","arxiv_id":"2301.12711","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-granularity-of-prosodic-representations-in","title":"On granularity of prosodic representations in expressive text-to-speech","date":"2023-01-26","arxiv_id":"2301.11446","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-low-resource-accents-without-accent","title":"Modelling low-resource accents without accent-specific TTS frontend","date":"2023-01-11","arxiv_id":"2301.04606","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifyspeech-a-unified-framework-for-zero-shot","title":"UnifySpeech: A Unified Framework for Zero-shot Text-to-Speech and Voice Conversion","date":"2023-01-10","arxiv_id":"2301.03801","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-automated-machine-translation-to","title":"Applying Automated Machine Translation to Educational Video Courses","date":"2023-01-09","arxiv_id":"2301.03141","repositories_listed":0,"syntology":null}],"record_sha256":"8cc2a40c0dd98d625b4f194f5710cadd6379c8b04b4597ceec9d3301ed6e8a8a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}