{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-1/papers/7","list_of":"/task/text-to-speech-1","task":"text-to-speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":15,"rows_per_page":100,"rows":[601,700],"of":1413,"counts":{"archive_papers_tagged":1413,"with_a_code_link":395,"where_syntology_ran_a_sample":106,"not_listed_spam_title":0,"listed":1413,"listed_where_code_ran":106,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-1","prev":"/task/text-to-speech-1/papers/6","next":"/task/text-to-speech-1/papers/8","papers":[{"url":null,"slug":"low-frame-rate-speech-codec-a-codec-designed","title":"Low Frame-rate Speech Codec: a Codec Designed for Fast High-quality Speech LLM Training and Inference","date":"2024-09-18","arxiv_id":"2409.12117","repositories_listed":0,"syntology":null},{"url":null,"slug":"spoofceleb-speech-deepfake-detection-and-sasv","title":"SpoofCeleb: Speech Deepfake Detection and SASV In The Wild","date":"2024-09-18","arxiv_id":"2409.17285","repositories_listed":0,"syntology":null},{"url":null,"slug":"spmis-an-investigation-of-synthetic-spoken","title":"SpMis: An Investigation of Synthetic Spoken Misinformation Detection","date":"2024-09-17","arxiv_id":"2409.11308","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-art-of-storytelling-multi-agent","title":"The Art of Storytelling: Multi-Agent Generative AI for Dynamic Multimodal Narratives","date":"2024-09-17","arxiv_id":"2409.11261","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-augmentation-for","title":"Zero Shot Text to Speech Augmentation for Automatic Speech Recognition on Low-Resource Accented Speech Corpora","date":"2024-09-17","arxiv_id":"2409.11107","repositories_listed":0,"syntology":null},{"url":null,"slug":"emo-dpo-controllable-emotional-speech","title":"Emo-DPO: Controllable Emotional Speech Synthesis through Direct Preference Optimization","date":"2024-09-16","arxiv_id":"2409.10157","repositories_listed":0,"syntology":null},{"url":null,"slug":"styletts-zs-efficient-high-quality-zero-shot","title":"StyleTTS-ZS: Efficient High-Quality Zero-Shot Text-to-Speech Synthesis with Distilled Time-Varying Style Diffusion","date":"2024-09-16","arxiv_id":"2409.10058","repositories_listed":0,"syntology":null},{"url":null,"slug":"acquiring-pronunciation-knowledge-from","title":"Acquiring Pronunciation Knowledge from Transcribed Speech Audio via Multi-task Learning","date":"2024-09-15","arxiv_id":"2409.09891","repositories_listed":0,"syntology":null},{"url":null,"slug":"e1-tts-simple-and-fast-non-autoregressive-tts","title":"E1 TTS: Simple and Fast Non-Autoregressive TTS","date":"2024-09-14","arxiv_id":"2409.09351","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-robustness-of-diffusion-based-zero","title":"Improving Robustness of Diffusion-Based Zero-Shot Speech Synthesis via Stable Formant Generation","date":"2024-09-14","arxiv_id":"2409.09311","repositories_listed":0,"syntology":null},{"url":null,"slug":"accentbox-towards-high-fidelity-zero-shot","title":"AccentBox: Towards High-Fidelity Zero-Shot Accent Generation","date":"2024-09-13","arxiv_id":"2409.09098","repositories_listed":0,"syntology":null},{"url":null,"slug":"hltcoe-jhu-submission-to-the-voice-privacy","title":"HLTCOE JHU Submission to the Voice Privacy Challenge 2024","date":"2024-09-13","arxiv_id":"2409.08913","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-in-the-wild","title":"Text-To-Speech Synthesis In The Wild","date":"2024-09-13","arxiv_id":"2409.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"full-text-error-correction-for-chinese-speech","title":"Full-text Error Correction for Chinese Speech Recognition with Large Language Model","date":"2024-09-12","arxiv_id":"2409.07790","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-dialect-text-to-speech-in-pitch-accent","title":"Cross-Dialect Text-To-Speech in Pitch-Accent Language Incorporating Multi-Dialect Phoneme-Level BERT","date":"2024-09-11","arxiv_id":"2409.07265","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-captcha-a-study-of-resilience-of-deepfake","title":"D-CAPTCHA++: A Study of Resilience of Deepfake CAPTCHA under Transferable Imperceptible Adversarial Attack","date":"2024-09-11","arxiv_id":"2409.07390","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-text-to-speech-as-golden-speech","title":"Zero-Shot Text-to-Speech as Golden Speech Generator: A Systematic Framework and its Applicability in Automatic Pronunciation Assessment","date":"2024-09-11","arxiv_id":"2409.07151","repositories_listed":0,"syntology":null},{"url":null,"slug":"2409-13734","title":"Enhancing Kurdish Text-to-Speech with Native Corpus Training: A High-Quality WaveGlow Vocoder Approach","date":"2024-09-10","arxiv_id":"2409.13734","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicewukong-benchmarking-deepfake-voice","title":"VoiceWukong: Benchmarking Deepfake Voice Detection","date":"2024-09-10","arxiv_id":"2409.06348","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-happens-to-diffusion-model-likelihood","title":"What happens to diffusion model likelihood when your model is conditional?","date":"2024-09-10","arxiv_id":"2409.06364","repositories_listed":0,"syntology":null},{"url":null,"slug":"as-speech-adaptive-style-for-speech-synthesis","title":"AS-Speech: Adaptive Style For Speech Synthesis","date":"2024-09-09","arxiv_id":"2409.05730","repositories_listed":0,"syntology":null},{"url":"/paper/last-language-model-aware-speech-tokenization","slug":"last-language-model-aware-speech-tokenization","title":"LAST: Language Model Aware Speech Tokenization","date":"2024-09-05","arxiv_id":"2409.03701","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-universal-vocoders-with-feature","title":"Training Universal Vocoders with Feature Smoothing-Based Augmentation Methods for High-Quality TTS Systems","date":"2024-09-04","arxiv_id":"2409.02517","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxhakka-a-dialectally-diverse-multi-speaker","title":"VoxHakka: A Dialectally Diverse Multi-speaker Text-to-Speech System for Taiwanese Hakka","date":"2024-09-03","arxiv_id":"2409.01548","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-synthetic-audio-conversations","title":"A Framework for Synthetic Audio Conversations Generation using Large Language Models","date":"2024-09-02","arxiv_id":"2409.00946","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multilingual-training-strategy-for-low","title":"A multilingual training strategy for low resource Text to Speech","date":"2024-09-02","arxiv_id":"2409.01217","repositories_listed":0,"syntology":null},{"url":null,"slug":"aasist3-kan-enhanced-aasist-speech-deepfake","title":"AASIST3: KAN-Enhanced AASIST Speech Deepfake Detection using SSL Features and Additional Regularization for the ASVspoof 2024 Challenge","date":"2024-08-30","arxiv_id":"2408.17352","repositories_listed":0,"syntology":null},{"url":null,"slug":"selecttts-synthesizing-anyone-s-voice-via","title":"SelectTTS: Synthesizing Anyone's Voice via Discrete Unit-Based Frame Selection","date":"2024-08-30","arxiv_id":"2408.17432","repositories_listed":0,"syntology":null},{"url":null,"slug":"easy-interpretable-effective-opensmile-for","title":"Easy, Interpretable, Effective: openSMILE for voice deepfake detection","date":"2024-08-28","arxiv_id":"2408.15775","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-adversarial-training-for-zero","title":"Multi-modal Adversarial Training for Zero-Shot Voice Cloning","date":"2024-08-28","arxiv_id":"2408.15916","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualspeech-enhancing-speaker-fidelity-and","title":"DualSpeech: Enhancing Speaker-Fidelity and Text-Intelligibility Through Dual Classifier-Free Guidance","date":"2024-08-26","arxiv_id":"2408.14423","repositories_listed":0,"syntology":null},{"url":null,"slug":"simplespeech-2-towards-simple-and-efficient","title":"SimpleSpeech 2: Towards Simple and Efficient Text-to-Speech with Flow-based Scalar Latent Transformer Diffusion Models","date":"2024-08-25","arxiv_id":"2408.13893","repositories_listed":0,"syntology":null},{"url":null,"slug":"positional-description-for-numerical","title":"Positional Description for Numerical Normalization","date":"2024-08-22","arxiv_id":"2408.12430","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-training-of-keyword-spotting-to","title":"Adversarial training of Keyword Spotting to Minimize TTS Data Overfitting","date":"2024-08-20","arxiv_id":"2408.10463","repositories_listed":0,"syntology":null},{"url":null,"slug":"ssl-tts-leveraging-self-supervised-embeddings","title":"kNN Retrieval for Simple and Effective Zero-Shot Multi-speaker Text-to-Speech","date":"2024-08-20","arxiv_id":"2408.10771","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-talker-finetuning-audio-language-model","title":"Style-Talker: Finetuning Audio Language Model and Style-Based Text-to-Speech Model for Fast Spoken Dialogue Generation","date":"2024-08-13","arxiv_id":"2408.11849","repositories_listed":0,"syntology":null},{"url":null,"slug":"fleurs-r-a-restored-multilingual-speech","title":"FLEURS-R: A Restored Multilingual Speech Corpus for Generation Tasks","date":"2024-08-12","arxiv_id":"2408.06227","repositories_listed":0,"syntology":null},{"url":null,"slug":"vq-ctap-cross-modal-fine-grained-sequence","title":"VQ-CTAP: Cross-Modal Fine-Grained Sequence Representation Learning for Speech Processing","date":"2024-08-11","arxiv_id":"2408.05758","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00284","title":"Bailing-TTS: Chinese Dialectal Speech Synthesis Towards Human-like Spontaneous Representation","date":"2024-08-01","arxiv_id":"2408.00284","repositories_listed":0,"syntology":null},{"url":null,"slug":"2407-21476","title":"On the Problem of Text-To-Speech Model Selection for Synthetic Data Generation in Automatic Speech Recognition","date":"2024-07-31","arxiv_id":"2407.21476","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-bandwidth-expansion-via-high-fidelity","title":"Speech Bandwidth Expansion Via High Fidelity Generative Adversarial Networks","date":"2024-07-26","arxiv_id":"2407.18571","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effect-of-purely-synthetic-training","title":"On the Effect of Purely Synthetic Training Data for Different Automatic Speech Recognition Architectures","date":"2024-07-25","arxiv_id":"2407.17997","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-vs-few-shot-multi-speaker-tts-using","title":"Zero-Shot vs. Few-Shot Multi-Speaker TTS Using Pre-trained Czech SpeechT5 Model","date":"2024-07-24","arxiv_id":"2407.17167","repositories_listed":0,"syntology":null},{"url":null,"slug":"synth4kws-synthesized-speech-for-user-defined","title":"Synth4Kws: Synthesized Speech for User Defined Keyword Spotting in Low Resource Environments","date":"2024-07-23","arxiv_id":"2407.16840","repositories_listed":0,"syntology":null},{"url":null,"slug":"braille-to-speech-generator-audio-generation","title":"Braille-to-Speech Generator: Audio Generation Based on Joint Fine-Tuning of CLIP and Fastspeech2","date":"2024-07-19","arxiv_id":"2407.14212","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-00004","title":"Handling Numeric Expressions in Automatic Speech Recognition","date":"2024-07-18","arxiv_id":"2408.00004","repositories_listed":0,"syntology":null},{"url":null,"slug":"spontaneous-style-text-to-speech-synthesis","title":"Spontaneous Style Text-to-Speech Synthesis with Controllable Spontaneous Behaviors Based on Language Models","date":"2024-07-18","arxiv_id":"2407.13509","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-language-modeling-approach-to-diacritic","title":"A Language Modeling Approach to Diacritic-Free Hebrew TTS","date":"2024-07-16","arxiv_id":"2407.12206","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-speech-synthesis-without","title":"Autoregressive Speech Synthesis without Vector Quantization","date":"2024-07-11","arxiv_id":"2407.08551","repositories_listed":0,"syntology":null},{"url":null,"slug":"source-tracing-of-audio-deepfake-systems","title":"Source Tracing of Audio Deepfake Systems","date":"2024-07-10","arxiv_id":"2407.08016","repositories_listed":0,"syntology":null},{"url":null,"slug":"asrrl-tts-agile-speaker-representation","title":"ASRRL-TTS: Agile Speaker Representation Reinforcement Learning for Text-to-Speech Speaker Adaptation","date":"2024-07-07","arxiv_id":"2407.05421","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-accented-speech-recognition-using","title":"Improving Accented Speech Recognition using Data Augmentation based on Unsupervised Text-to-Speech Synthesis","date":"2024-07-04","arxiv_id":"2407.04047","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-acoustic-bpe-in","title":"On the Effectiveness of Acoustic BPE in Decoder-Only TTS","date":"2024-07-04","arxiv_id":"2407.03892","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-zero-shot-text-to-speech-synthesis","title":"Robust Zero-Shot Text-to-Speech Synthesis with Reverse Inference Optimization","date":"2024-07-02","arxiv_id":"2407.02243","repositories_listed":0,"syntology":null},{"url":null,"slug":"ttslow-slow-down-text-to-speech-with","title":"TTSlow: Slow Down Text-to-Speech with Efficiency Robustness Evaluations","date":"2024-07-02","arxiv_id":"2407.01927","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-zero-shot-text-to-speech-with","title":"Lightweight Zero-shot Text-to-Speech with Mixture of Adapters","date":"2024-07-01","arxiv_id":"2407.01291","repositories_listed":0,"syntology":null},{"url":null,"slug":"fly-tts-fast-lightweight-and-high-quality-end","title":"FLY-TTS: Fast, Lightweight and High-Quality End-to-End Text-to-Speech Synthesis","date":"2024-06-30","arxiv_id":"2407.00753","repositories_listed":0,"syntology":null},{"url":null,"slug":"naist-simultaneous-speech-translation-system","title":"NAIST Simultaneous Speech Translation System for IWSLT 2024","date":"2024-06-30","arxiv_id":"2407.00826","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-source-conversational-ai-with","title":"Open-Source Conversational AI with SpeechBrain 1.0","date":"2024-06-29","arxiv_id":"2407.00463","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-of-asv-for-voice-identification","title":"Application of ASV for Voice Identification after VC and Duration Predictor Improvement in TTS Models","date":"2024-06-27","arxiv_id":"2406.19243","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-for-hindi","title":"Automatic Speech Recognition for Hindi","date":"2024-06-26","arxiv_id":"2406.18135","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-driven-multimodal-opinion-expression","title":"LLM-Driven Multimodal Opinion Expression Identification","date":"2024-06-26","arxiv_id":"2406.18088","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-fidelity-text-to-speech-via-discrete","title":"High Fidelity Text-to-Speech Via Discrete Tokens Using Token Transducer and Group Masked Language Model","date":"2024-06-25","arxiv_id":"2406.17310","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-robustness-of-llm-based-speech","title":"Improving Robustness of LLM-based Speech Synthesis by Learning Monotonic Alignment","date":"2024-06-25","arxiv_id":"2406.17957","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-parameter-efficient-transfer","title":"Leveraging Parameter-Efficient Transfer Learning for Multi-Lingual Text-to-Speech Adaptation","date":"2024-06-25","arxiv_id":"2406.17257","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-text-to-speech-for-arabic","title":"Towards Zero-Shot Text-To-Speech for Arabic Dialects","date":"2024-06-24","arxiv_id":"2406.16751","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-speaker-multi-lingual-voice-cloning","title":"A multi-speaker multi-lingual voice cloning system based on vits2 for limmits 2024 challenge","date":"2024-06-22","arxiv_id":"2406.17801","repositories_listed":0,"syntology":null},{"url":null,"slug":"interbiasing-boost-unseen-word-recognition","title":"InterBiasing: Boost Unseen Word Recognition through Biasing Intermediate Predictions","date":"2024-06-21","arxiv_id":"2406.14890","repositories_listed":0,"syntology":null},{"url":null,"slug":"dasb-discrete-audio-and-speech-benchmark","title":"DASB -- Discrete Audio and Speech Benchmark","date":"2024-06-20","arxiv_id":"2406.14294","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-data-generation-and-unsupervised","title":"Instruction Data Generation and Unsupervised Adaptation for Speech Language Models","date":"2024-06-18","arxiv_id":"2406.12946","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-accent-modeling-with","title":"Multi-Scale Accent Modeling and Disentangling for Multi-Speaker Multi-Accent Text-to-Speech Synthesis","date":"2024-06-16","arxiv_id":"2406.10844","repositories_listed":0,"syntology":null},{"url":null,"slug":"phoneme-discretized-saliency-maps-for","title":"Phoneme Discretized Saliency Maps for Explainable Detection of AI-Generated Voice","date":"2024-06-14","arxiv_id":"2406.10422","repositories_listed":0,"syntology":null},{"url":null,"slug":"disfluencyspeech-single-speaker","title":"DisfluencySpeech -- Single-Speaker Conversational Speech Dataset with Paralanguage","date":"2024-06-13","arxiv_id":"2406.08820","repositories_listed":0,"syntology":null},{"url":null,"slug":"dubwise-video-guided-speech-duration-control","title":"DubWise: Video-Guided Speech Duration Control in Multimodal LLM-based Text-to-Speech for Dubbing","date":"2024-06-13","arxiv_id":"2406.08802","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-conditioned-phonemic-and-prosodic","title":"Audio-conditioned phonemic and prosodic annotation for building text-to-speech models from unlabeled speech data","date":"2024-06-12","arxiv_id":"2406.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-e-r-robust-and-efficient-zero-shot-text","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","date":"2024-06-12","arxiv_id":"2406.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"vecl-tts-voice-identity-and-emotional-style","title":"VECL-TTS: Voice identity and Emotional style controllable Cross-Lingual Text-to-Speech","date":"2024-06-12","arxiv_id":"2406.08076","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-achieve-high-quality-direct-speech-to","title":"Can We Achieve High-quality Direct Speech-to-Speech Translation without Parallel Speech Data?","date":"2024-06-11","arxiv_id":"2406.07289","repositories_listed":0,"syntology":null},{"url":null,"slug":"makesinger-a-semi-supervised-training-method","title":"MakeSinger: A Semi-Supervised Training Method for Data-Efficient Singing Voice Synthesis via Classifier-free Diffusion Guidance","date":"2024-06-10","arxiv_id":"2406.05965","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-noise-robustness-for-flow","title":"An Investigation of Noise Robustness for Flow-Matching-Based Zero-Shot TTS","date":"2024-06-09","arxiv_id":"2406.05699","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-aware-and-context-aware-expressive","title":"Text-aware and Context-aware Expressive Audiobook Speech Synthesis","date":"2024-06-09","arxiv_id":"2406.05672","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-diffusion-transformer-for-text","title":"Autoregressive Diffusion Transformer for Text-to-Speech Synthesis","date":"2024-06-08","arxiv_id":"2406.05551","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-e-2-neural-codec-language-models-are","title":"VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers","date":"2024-06-08","arxiv_id":"2406.05370","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-diffusion-model-for-spectrogram-up","title":"Boosting Diffusion Model for Spectrogram Up-sampling in Text-to-speech: An Empirical Study","date":"2024-06-07","arxiv_id":"2406.04633","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-codecs-spectrogram-based-audio","title":"Spectral Codecs: Improving Non-Autoregressive Speech Synthesis with Spectrogram-Based Audio Codecs","date":"2024-06-07","arxiv_id":"2406.05298","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-human-in-the-loop-approach-to-improving","title":"A Human-in-the-Loop Approach to Improving Cross-Text Prosody Transfer","date":"2024-06-06","arxiv_id":"2406.06601","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-audio-codec-based-zero-shot-text-to","title":"Improving Audio Codec-based Zero-Shot Text-to-Speech Synthesis with Multi-Modal Context and Large Language Model","date":"2024-06-06","arxiv_id":"2406.03706","repositories_listed":0,"syntology":null},{"url":null,"slug":"total-duration-aware-duration-modeling-for","title":"Total-Duration-Aware Duration Modeling for Text-to-Speech Systems","date":"2024-06-06","arxiv_id":"2406.04281","repositories_listed":0,"syntology":null},{"url":null,"slug":"harder-or-different-understanding","title":"Harder or Different? Understanding Generalization of Audio Deepfake Detection","date":"2024-06-05","arxiv_id":"2406.03512","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-mixture-of-experts-for-expressive-text","title":"Style Mixture of Experts for Expressive Text-To-Speech Synthesis","date":"2024-06-05","arxiv_id":"2406.03637","repositories_listed":0,"syntology":null},{"url":null,"slug":"syn2real-leveraging-task-arithmetic-for","title":"Task Arithmetic can Mitigate Synthetic-to-Real Gap in Automatic Speech Recognition","date":"2024-06-05","arxiv_id":"2406.02925","repositories_listed":0,"syntology":null},{"url":null,"slug":"bivocoder-a-bidirectional-neural-vocoder","title":"BiVocoder: A Bidirectional Neural Vocoder Integrating Feature Extraction and Waveform Generation","date":"2024-06-04","arxiv_id":"2406.02162","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-multimodal-transformers-with-a","title":"Discrete Multimodal Transformers with a Pretrained Large Language Model for Mixed-Supervision Speech Processing","date":"2024-06-04","arxiv_id":"2406.06582","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-enhanced-language-modeling-for-text","title":"Phonetic Enhanced Language Modeling for Text-to-Speech Synthesis","date":"2024-06-04","arxiv_id":"2406.02009","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-conversion-in-text-to-speech-using","title":"Accent Conversion in Text-To-Speech Using Multi-Level VAE and Adversarial Training","date":"2024-06-03","arxiv_id":"2406.01018","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-zero-shot-text-to-speech-synthesis","title":"Enhancing Zero-shot Text-to-Speech Synthesis with Human Feedback","date":"2024-06-02","arxiv_id":"2406.00654","repositories_listed":0,"syntology":null},{"url":null,"slug":"zipper-a-multi-tower-decoder-architecture-for","title":"Zipper: A Multi-Tower Decoder Architecture for Fusing Modalities","date":"2024-05-29","arxiv_id":"2405.18669","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-lm-pushing-the-limits-of-error","title":"Denoising LM: Pushing the Limits of Error Correction Models for Speech Recognition","date":"2024-05-24","arxiv_id":"2405.15216","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-prosody-transfer-comparing","title":"Multilingual Prosody Transfer: Comparing Supervised & Transfer Learning","date":"2024-05-23","arxiv_id":"2406.00022","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-fine-tuning-text-1","title":"DLPO: Diffusion Model Loss-Guided Reinforcement Learning for Fine-Tuning Text-to-Speech Diffusion Models","date":"2024-05-23","arxiv_id":"2405.14632","repositories_listed":0,"syntology":null}],"record_sha256":"b2c1cc23fa588052ac007ca0533a8c101672fc7f800b1822a36f63f9c4027d00","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}