{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-to-speech-1/papers/6","list_of":"/task/text-to-speech-1","task":"text-to-speech","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":15,"rows_per_page":100,"rows":[501,600],"of":1413,"counts":{"archive_papers_tagged":1413,"with_a_code_link":395,"where_syntology_ran_a_sample":106,"not_listed_spam_title":0,"listed":1413,"listed_where_code_ran":106,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":95,"every_run_a_failure_of_syntologys_instrument":11,"listed_with_a_run_with_no_instrument_failure":95,"listed_every_run_a_failure_of_syntologys_instrument":11,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-to-speech-1","prev":"/task/text-to-speech-1/papers/5","next":"/task/text-to-speech-1/papers/7","papers":[{"url":null,"slug":"fine-grained-preference-optimization-improves","title":"Fine-grained Preference Optimization Improves Zero-shot Text-to-Speech","date":"2025-02-05","arxiv_id":"2502.02950","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-speaker-change-detection-and-gender","title":"Streaming Speaker Change Detection and Gender Classification for Transducer-Based Multi-Talker Speech Translation","date":"2025-02-04","arxiv_id":"2502.02683","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotalkinggaussian-continuous-emotion","title":"EmoTalkingGaussian: Continuous Emotion-conditioned Talking Head Synthesis","date":"2025-02-02","arxiv_id":"2502.00654","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualspeech-enhance-prosody-with-visual","title":"VisualSpeech: Enhance Prosody with Visual Context in TTS","date":"2025-01-31","arxiv_id":"2501.19258","repositories_listed":0,"syntology":null},{"url":null,"slug":"breezyvoice-adapting-tts-for-taiwanese","title":"BreezyVoice: Adapting TTS for Taiwanese Mandarin with Enhanced Polyphone Disambiguation -- Challenges and Insights","date":"2025-01-29","arxiv_id":"2501.17790","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-neural-tts-voices-for-accessibility","title":"Compact Neural TTS Voices for Accessibility","date":"2025-01-28","arxiv_id":"2501.17332","repositories_listed":0,"syntology":null},{"url":null,"slug":"characteristic-specific-partial-fine-tuning","title":"Characteristic-Specific Partial Fine-Tuning for Efficient Emotion and Speaker Adaptation in Codec Language Text-to-Speech Models","date":"2025-01-24","arxiv_id":"2501.14273","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-audio-deepfake-detection-via","title":"Generalizable Audio Deepfake Detection via Latent Space Refinement and Augmentation","date":"2025-01-24","arxiv_id":"2501.14240","repositories_listed":0,"syntology":null},{"url":null,"slug":"locoml-a-framework-for-real-world-ml","title":"LoCoML: A Framework for Real-World ML Inference Pipelines","date":"2025-01-24","arxiv_id":"2501.14165","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-data-augmentation-challenge-zero","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","date":"2025-01-23","arxiv_id":"2501.13372","repositories_listed":0,"syntology":null},{"url":null,"slug":"development-of-an-inclusive-educational","title":"Development of an Inclusive Educational Platform Using Open Technologies and Machine Learning: A Case Study on Accessibility Enhancement","date":"2025-01-22","arxiv_id":"2503.15501","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-domain-adaptation-framework-for-speech","title":"A Domain Adaptation Framework for Speech Recognition Systems with Only Synthetic data","date":"2025-01-21","arxiv_id":"2501.12501","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-synthesis-along-perceptual-voice","title":"Speech Synthesis along Perceptual Voice Quality Dimensions","date":"2025-01-15","arxiv_id":"2501.08791","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-lightweight-and-stable-zero-shot-tts","title":"Towards Lightweight and Stable Zero-shot TTS with Self-distilled Representation Disentanglement","date":"2025-01-15","arxiv_id":"2501.08566","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-powered-assistive-technologies-for-visual","title":"AI-Powered Assistive Technologies for Visual Impairment","date":"2025-01-14","arxiv_id":"2503.15494","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-text-to-speech-synthesis-using","title":"Low-Resource Text-to-Speech Synthesis Using Noise-Augmented Training of ForwardTacotron","date":"2025-01-10","arxiv_id":"2501.05976","repositories_listed":0,"syntology":null},{"url":null,"slug":"mars6-a-small-and-robust-hierarchical-codec","title":"MARS6: A Small and Robust Hierarchical-Codec Text-to-Speech Model","date":"2025-01-10","arxiv_id":"2501.05787","repositories_listed":0,"syntology":null},{"url":null,"slug":"minmo-a-multimodal-large-language-model-for","title":"MinMo: A Multimodal Large Language Model for Seamless Voice Interaction","date":"2025-01-10","arxiv_id":"2501.06282","repositories_listed":0,"syntology":null},{"url":null,"slug":"proemo-prompt-driven-text-to-speech-synthesis","title":"PROEMO: Prompt-Driven Text-to-Speech Synthesis Based on Emotion and Intensity Control","date":"2025-01-10","arxiv_id":"2501.06276","repositories_listed":0,"syntology":null},{"url":null,"slug":"tts-transducer-end-to-end-speech-synthesis","title":"TTS-Transducer: End-to-End Speech Synthesis with Neural Transducer","date":"2025-01-10","arxiv_id":"2501.06320","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-speaker-specific-features-in-speaker","title":"Probing Speaker-specific Features in Speaker Representations","date":"2025-01-09","arxiv_id":"2501.05310","repositories_listed":0,"syntology":null},{"url":null,"slug":"cued-speech-generation-leveraging-a-pre","title":"Cued Speech Generation Leveraging a Pre-trained Audiovisual Text-to-Speech Model","date":"2025-01-08","arxiv_id":"2501.04799","repositories_listed":0,"syntology":null},{"url":null,"slug":"disambiguation-of-chinese-polyphones-in-an","title":"Disambiguation of Chinese Polyphones in an End-to-End Framework with Semantic Features Extracted by Pre-trained BERT","date":"2025-01-02","arxiv_id":"2501.01102","repositories_listed":0,"syntology":null},{"url":"/paper/facespeak-expressive-and-high-quality-speech","slug":"facespeak-expressive-and-high-quality-speech","title":"FaceSpeak: Expressive and High-Quality Speech Synthesis from Human Portraits of Different Styles","date":"2025-01-02","arxiv_id":"2501.03181","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/facespeak-expressive-and-high-quality-speech#ran","syntology_url":"https://syntology.ai/paper/2501.03181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03181"}},"official":null}},{"url":null,"slug":"stable-tts-stable-speaker-adaptive-text-to","title":"Stable-TTS: Stable Speaker-Adaptive Text-to-Speech Synthesis via Prosody Prompting","date":"2024-12-28","arxiv_id":"2412.20155","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-ve-heard-of-you-generate-spoken-named","title":"\"I've Heard of You!\": Generate Spoken Named Entity Recognition Data for Unseen Entities","date":"2024-12-26","arxiv_id":"2412.19102","repositories_listed":0,"syntology":null},{"url":null,"slug":"indonesian-english-code-switching-speech","title":"Indonesian-English Code-Switching Speech Synthesizer Utilizing Multilingual STEN-TTS and Bert LID","date":"2024-12-26","arxiv_id":"2412.19043","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-nam-to-speech-conversion-with-novel","title":"Advancing NAM-to-Speech Conversion with Novel Methods and the MultiNAM Dataset","date":"2024-12-25","arxiv_id":"2412.18839","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-speech-synthesis-with-next","title":"Autoregressive Speech Synthesis with Next-Distribution Prediction","date":"2024-12-22","arxiv_id":"2412.16846","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-disentanglement-for-environment","title":"Incremental Disentanglement for Environment-Aware Zero-Shot Text-to-Speech Synthesis","date":"2024-12-22","arxiv_id":"2412.16977","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-do-speech-language-models-fail-to","title":"Why Do Speech Language Models Fail to Generate Semantically Coherent Outputs? A Modality Evolving Perspective","date":"2024-12-22","arxiv_id":"2412.17048","repositories_listed":0,"syntology":null},{"url":null,"slug":"interleaved-speech-text-language-models-are","title":"Interleaved Speech-Text Language Models are Simple Streaming Text to Speech Synthesizers","date":"2024-12-20","arxiv_id":"2412.16102","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-this-not-that-investigating-key-dataset","title":"Scale This, Not That: Investigating Key Dataset Attributes for Efficient Speech Enhancement Scaling","date":"2024-12-19","arxiv_id":"2412.14890","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-naturalness-in-llm-generated","title":"Enhancing Naturalness in LLM-Generated Utterances through Disfluency Insertion","date":"2024-12-17","arxiv_id":"2412.12710","repositories_listed":0,"syntology":null},{"url":null,"slug":"phoneme-level-feature-discrepancies-a-key-to","title":"Phoneme-Level Feature Discrepancies: A Key to Detecting Sophisticated Speech Deepfakes","date":"2024-12-17","arxiv_id":"2412.12619","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosodyfm-unsupervised-phrasing-and","title":"ProsodyFM: Unsupervised Phrasing and Intonation Control for Intelligible Speech Synthesis","date":"2024-12-16","arxiv_id":"2412.11795","repositories_listed":0,"syntology":null},{"url":null,"slug":"amused-an-attentive-deep-neural-network-for","title":"AMuSeD: An Attentive Deep Neural Network for Multimodal Sarcasm Detection Incorporating Bi-modal Data Augmentation","date":"2024-12-13","arxiv_id":"2412.10103","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-generative-modeling-with-residual","title":"Efficient Generative Modeling with Residual Vector Quantization-Based Tokens","date":"2024-12-13","arxiv_id":"2412.10208","repositories_listed":0,"syntology":null},{"url":null,"slug":"cssinger-end-to-end-chunkwise-streaming","title":"CSSinger: End-to-End Chunkwise Streaming Singing Voice Synthesis System Based on Conditional Variational Autoencoder","date":"2024-12-12","arxiv_id":"2412.08918","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-analysis-of-automatic-word-and","title":"A Preliminary Analysis of Automatic Word and Syllable Prominence Detection in Non-Native Speech With Text-to-Speech Prosody Embeddings","date":"2024-12-11","arxiv_id":"2412.08283","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-model-for-voice-and-accent","title":"A Unified Model For Voice and Accent Conversion In Speech and Singing using Self-Supervised Learning and Feature Extraction","date":"2024-12-11","arxiv_id":"2412.08312","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligner-guided-training-paradigm-advancing","title":"Aligner-Guided Training Paradigm: Advancing Text-to-Speech Models with Aligner Guided Duration","date":"2024-12-11","arxiv_id":"2412.08112","repositories_listed":0,"syntology":null},{"url":null,"slug":"latentspeech-latent-diffusion-for-text-to","title":"LatentSpeech: Latent Diffusion for Text-To-Speech Generation","date":"2024-12-11","arxiv_id":"2412.08117","repositories_listed":0,"syntology":null},{"url":null,"slug":"emospeech-a-corpus-of-emotionally-rich-and","title":"EmoSpeech: A Corpus of Emotionally Rich and Contextually Detailed Speech Annotations","date":"2024-12-09","arxiv_id":"2412.06581","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffstyletts-diffusion-based-hierarchical","title":"DiffStyleTTS: Diffusion-based Hierarchical Prosody Modeling for Text-to-Speech with Diverse and Controllable Styles","date":"2024-12-04","arxiv_id":"2412.03388","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-is-not-all-you-need-multimodal-prompting","title":"Text Is Not All You Need: Multimodal Prompting Helps LLMs Understand Humor","date":"2024-12-01","arxiv_id":"2412.05315","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-in-machine-speech-chain","title":"Continual Learning in Machine Speech Chain Using Gradient Episodic Memory","date":"2024-11-27","arxiv_id":"2411.18320","repositories_listed":0,"syntology":null},{"url":null,"slug":"salmonn-omni-a-codec-free-llm-for-full-duplex","title":"SALMONN-omni: A Codec-free LLM for Full-duplex Speech Understanding and Generation","date":"2024-11-27","arxiv_id":"2411.18138","repositories_listed":0,"syntology":null},{"url":null,"slug":"visatronic-a-multimodal-decoder-only-model","title":"Visatronic: A Multimodal Decoder-Only Model for Speech Synthesis","date":"2024-11-26","arxiv_id":"2411.17690","repositories_listed":0,"syntology":null},{"url":null,"slug":"hard-synth-synthesizing-diverse-hard-samples","title":"Hard-Synth: Synthesizing Diverse Hard Samples for ASR using Zero-Shot TTS and LLM","date":"2024-11-20","arxiv_id":"2411.13159","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-context-based-numerical-format-prediction","title":"A Context-Based Numerical Format Prediction for a Text-To-Speech System","date":"2024-11-19","arxiv_id":"2412.00028","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-virtual-reality-and-ai-tutoring","title":"Leveraging Virtual Reality and AI Tutoring for Language Learning: A Case Study of a Virtual Campus Environment with OpenAI GPT Integration with Unity 3D","date":"2024-11-19","arxiv_id":"2411.12619","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-mushra-addressing-modern","title":"Rethinking MUSHRA: Addressing Modern Challenges in Text-to-Speech Evaluation","date":"2024-11-19","arxiv_id":"2411.12719","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-grapheme-to-phoneme-conversion","title":"Improving Grapheme-to-Phoneme Conversion through In-Context Knowledge Retrieval with Large Language Models","date":"2024-11-12","arxiv_id":"2411.07563","repositories_listed":0,"syntology":null},{"url":null,"slug":"debatts-zero-shot-debating-text-to-speech","title":"Debatts: Zero-Shot Debating Text-to-Speech Synthesis","date":"2024-11-10","arxiv_id":"2411.06540","repositories_listed":0,"syntology":null},{"url":null,"slug":"cuify-the-xr-an-open-source-package-to-embed","title":"CUIfy the XR: An Open-Source Package to Embed LLM-powered Conversational Agents in XR","date":"2024-11-07","arxiv_id":"2411.04671","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-is-more-than-words-do-speech-to-text","title":"Speech is More Than Words: Do Speech-to-Text Translation Systems Leverage Prosody?","date":"2024-10-31","arxiv_id":"2410.24019","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-high-quality-auto-regressive-speech","title":"Fast and High-Quality Auto-Regressive Speech Synthesis via Speculative Decoding","date":"2024-10-29","arxiv_id":"2410.21951","repositories_listed":0,"syntology":null},{"url":null,"slug":"rdsinger-reference-based-diffusion-network","title":"RDSinger: Reference-based Diffusion Network for Singing Voice Synthesis","date":"2024-10-29","arxiv_id":"2410.21641","repositories_listed":0,"syntology":null},{"url":null,"slug":"asynchronous-tool-usage-for-real-time-agents","title":"Asynchronous Tool Usage for Real-Time Agents","date":"2024-10-28","arxiv_id":"2410.21620","repositories_listed":0,"syntology":null},{"url":null,"slug":"get-large-language-models-ready-to-speak-a","title":"Get Large Language Models Ready to Speak: A Late-fusion Approach for Speech Generation","date":"2024-10-27","arxiv_id":"2410.20336","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-improving-automatic-speech","title":"Evaluating and Improving Automatic Speech Recognition Systems for Korean Meteorological Experts","date":"2024-10-24","arxiv_id":"2410.18444","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-social-platforms-accessible-emotion","title":"Making Social Platforms Accessible: Emotion-Aware Speech Generation with Integrated Text Analysis","date":"2024-10-24","arxiv_id":"2410.19199","repositories_listed":0,"syntology":null},{"url":null,"slug":"elaichi-enhancing-low-resource-tts-by","title":"ELAICHI: Enhancing Low-resource TTS by Addressing Infrequent and Low-frequency Character Bigrams","date":"2024-10-23","arxiv_id":"2410.17901","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-low-resource-asr-through-versatile","title":"Enhancing Low-Resource ASR through Versatile TTS: Bridging the Data Gap","date":"2024-10-22","arxiv_id":"2410.16726","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-speech-synthesis-using-per-token","title":"Continuous Speech Synthesis using per-token Latent Diffusion","date":"2024-10-21","arxiv_id":"2410.16048","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-collecting-text-to","title":"A Unified Framework for Collecting Text-to-Speech Synthesis Datasets for 22 Indian Languages","date":"2024-10-18","arxiv_id":"2410.14197","repositories_listed":0,"syntology":null},{"url":null,"slug":"dart-disentanglement-of-accent-and-speaker","title":"DART: Disentanglement of Accent and Speaker Representation in Multispeaker Text-to-Speech","date":"2024-10-17","arxiv_id":"2410.13342","repositories_listed":0,"syntology":null},{"url":null,"slug":"durian-e-2-duration-informed-attention","title":"DurIAN-E 2: Duration Informed Attention Network with Adaptive Variational Autoencoder and Adversarial Learning for Expressive Text-to-Speech Synthesis","date":"2024-10-17","arxiv_id":"2410.13288","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-crowdsourced-audio-for-text-to","title":"Enhancing Crowdsourced Audio for Text-to-Speech Models","date":"2024-10-17","arxiv_id":"2410.13357","repositories_listed":0,"syntology":null},{"url":null,"slug":"failing-forward-improving-generative-error","title":"Failing Forward: Improving Generative Error Correction for ASR with Synthetic Data and Retrieval Augmentation","date":"2024-10-17","arxiv_id":"2410.13198","repositories_listed":0,"syntology":null},{"url":null,"slug":"ervq-enhanced-residual-vector-quantization","title":"ERVQ: Enhanced Residual Vector Quantization with Intra-and-Inter-Codebook Optimization for Neural Audio Codecs","date":"2024-10-16","arxiv_id":"2410.12359","repositories_listed":0,"syntology":null},{"url":null,"slug":"dmdspeech-distilled-diffusion-model","title":"DMOSpeech: Direct Metric Optimization via Distilled Diffusion Model in Zero-Shot Speech Synthesis","date":"2024-10-14","arxiv_id":"2410.11097","repositories_listed":0,"syntology":null},{"url":null,"slug":"isochronometer-a-simple-and-effective","title":"IsoChronoMeter: A simple and effective isochronic translation evaluation metric","date":"2024-10-14","arxiv_id":"2410.11127","repositories_listed":0,"syntology":null},{"url":null,"slug":"emphasis-rendering-for-conversational-text-to","title":"Emphasis Rendering for Conversational Text-to-Speech with Multi-modal Multi-scale Context Modeling","date":"2024-10-12","arxiv_id":"2410.09524","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-data-validation-methods-for","title":"Unsupervised Data Validation Methods for Efficient Model Training","date":"2024-10-10","arxiv_id":"2410.07880","repositories_listed":0,"syntology":null},{"url":null,"slug":"bahasa-harmony-a-comprehensive-dataset-for","title":"Bahasa Harmony: A Comprehensive Dataset for Bahasa Text-to-Speech Synthesis with Discrete Codec Modeling of EnGen-TTS","date":"2024-10-09","arxiv_id":"2410.06608","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-deepfake-speech-be-reliably-detected","title":"Can DeepFake Speech be Reliably Detected?","date":"2024-10-09","arxiv_id":"2410.06572","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-training-strategies-for-natural","title":"Efficient training strategies for natural sounding speech synthesis and speaker adaptation based on FastPitch","date":"2024-10-09","arxiv_id":"2410.06787","repositories_listed":0,"syntology":null},{"url":null,"slug":"seginr-segment-wise-implicit-neural","title":"SegINR: Segment-wise Implicit Neural Representation for Sequence Alignment in Neural Text-to-Speech","date":"2024-10-07","arxiv_id":"2410.04690","repositories_listed":0,"syntology":null},{"url":null,"slug":"hall-e-hierarchical-neural-codec-language","title":"HALL-E: Hierarchical Neural Codec Language Model for Minute-Long Zero-Shot Text-to-Speech Synthesis","date":"2024-10-06","arxiv_id":"2410.04380","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-and-robust-defenses-in","title":"Adversarial Attacks and Robust Defenses in Speaker Embedding based Zero-Shot Text-to-Speech System","date":"2024-10-05","arxiv_id":"2410.04017","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-semantic-communication-for-text-to","title":"Generative Semantic Communication for Text-to-Speech Synthesis","date":"2024-10-04","arxiv_id":"2410.03459","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiverse-efficient-and-expressive-zero-shot","title":"MultiVerse: Efficient and Expressive Zero-Shot Multi-Task Text-to-Speech","date":"2024-10-04","arxiv_id":"2410.03192","repositories_listed":0,"syntology":null},{"url":null,"slug":"textless-streaming-speech-to-speech","title":"Textless Streaming Speech-to-Speech Translation using Semantic Speech Tokens","date":"2024-10-04","arxiv_id":"2410.03298","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmentation-through-laundering-attacks-for","title":"Augmentation through Laundering Attacks for Audio Spoof Detection","date":"2024-10-01","arxiv_id":"2410.01108","repositories_listed":0,"syntology":null},{"url":null,"slug":"accent-conversion-using-discrete-units-with","title":"Accent conversion using discrete units with parallel data synthesized from controllable accented TTS","date":"2024-09-30","arxiv_id":"2410.03734","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-wise-intonation-model-for-cross-language","title":"Word-wise intonation model for cross-language TTS systems","date":"2024-09-30","arxiv_id":"2409.20374","repositories_listed":0,"syntology":null},{"url":null,"slug":"description-based-controllable-text-to-speech","title":"Description-based Controllable Text-to-Speech with Cross-Lingual Voice Control","date":"2024-09-26","arxiv_id":"2409.17452","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotional-dimension-control-in-language-model","title":"Emotional Dimension Control in Language Model-Based Text-to-Speech: Spanning a Broad Spectrum of Human Emotions","date":"2024-09-25","arxiv_id":"2409.16681","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-synthetic-data-for-cross-speaker","title":"Exploring synthetic data for cross-speaker style transfer in style representation based TTS","date":"2024-09-25","arxiv_id":"2409.17364","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-text-to-text-an-overview-of-multimodal","title":"Beyond Text-to-Text: An Overview of Multimodal and Generative Artificial Intelligence for Education Using Topic Modeling","date":"2024-09-24","arxiv_id":"2409.16376","repositories_listed":0,"syntology":null},{"url":null,"slug":"facial-expression-enhanced-tts-combining-face","title":"Facial Expression-Enhanced TTS: Combining Face Representation and Emotion Intensity for Adaptive Speech","date":"2024-09-24","arxiv_id":"2409.16203","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylefusion-tts-multimodal-style-control-and","title":"StyleFusion TTS: Multimodal Style-control and Enhanced Feature Fusion for Zero-shot Text-to-speech Synthesis","date":"2024-09-24","arxiv_id":"2409.15741","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-feasibility-of-fully-ai-automated","title":"On the Feasibility of Fully AI-automated Vishing Attacks","date":"2024-09-20","arxiv_id":"2409.13793","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-cross-lingual-voice-transfer-for","title":"Zero-shot Cross-lingual Voice Transfer for TTS","date":"2024-09-20","arxiv_id":"2409.13910","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-synthetic-training-data-for-speech","title":"Enhancing Synthetic Training Data for Speech Commands: From ASR-Based Filtering to Domain Adaptation in SSL Latent Space","date":"2024-09-19","arxiv_id":"2409.12745","repositories_listed":0,"syntology":null},{"url":null,"slug":"preference-alignment-improves-language-model","title":"Preference Alignment Improves Language Model-Based TTS","date":"2024-09-19","arxiv_id":"2409.12403","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpi-tts-directional-patch-interaction-for","title":"DPI-TTS: Directional Patch Interaction for Fast-Converging and Style Temporal Modeling in Text-to-Speech","date":"2024-09-18","arxiv_id":"2409.11835","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-an-inter-pausal-unit-ipu-based","title":"Exploring an Inter-Pausal Unit (IPU) based Approach for Indic End-to-End TTS Systems","date":"2024-09-18","arxiv_id":"2409.11915","repositories_listed":0,"syntology":null}],"record_sha256":"4d85cc79bf72d8c37c81dbca03c0a3867d09452b825a723bf7dc0a400f155662","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}