{"url":"/task/voice-cloning","name":"Voice Cloning","slug":"voice-cloning","description_markdown":"Voice cloning is a highly desired feature for personalized speech interfaces. Neural voice cloning system learns to synthesize a person’s voice from only a few audio samples.","categories":[{"name":"Adversarial","url":"/area/adversarial"},{"name":"Audio","url":"/area/audio"},{"name":"Computer Code","url":"/area/computer-code"},{"name":"Computer Vision","url":"/area/computer-vision"},{"name":"Medical","url":"/area/medical"},{"name":"Methodology","url":"/area/methodology"},{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Music","url":"/area/music"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Reasoning","url":"/area/reasoning"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":112,"papers_with_code":36,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[{"url":"/dataset/gneutralspeech-female","name":"GneutralSpeech Female","full_name":"","num_papers_in_archive":1},{"url":"/dataset/gneutralspeech-male","name":"GneutralSpeech Male","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/audio-generation","name":"Audio Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":36,"tagged_in_all":112,"items":[{"url":"/paper/transfer-learning-from-speaker-verification","title":"Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis","date":"2018-06-12","arxiv_id":"1806.04558","repositories_listed":11,"syntology":null},{"url":"/paper/learning-to-speak-fluently-in-a-foreign","title":"Learning to Speak Fluently in a Foreign Language: Multilingual Speech Synthesis and Cross-Language Voice Cloning","date":"2019-07-09","arxiv_id":"1907.04448","repositories_listed":4,"syntology":{"n":9,"n_ran":2,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/funaudiollm-voice-understanding-and","title":"FunAudioLLM: Voice Understanding and Generation Foundation Models for Natural Interaction Between Humans and LLMs","date":"2024-07-04","arxiv_id":"2407.04051","repositories_listed":3,"syntology":{"n":10,"n_ran":10,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/enhancing-suno-s-bark-text-to-speech-model","title":"Enhancing Suno's Bark Text-to-Speech Model: Addressing Limitations Through Meta's Encodec and Pre-Trained Hubert","date":"2023-04-18","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/wavlm-model-ensemble-for-audio-deepfake","title":"WavLM model ensemble for audio deepfake detection","date":"2024-08-14","arxiv_id":"2408.07414","repositories_listed":2,"syntology":null},{"url":"/paper/anonymizing-speech-evaluating-and-designing","title":"Anonymizing Speech: Evaluating and Designing Speaker Anonymization Techniques","date":"2023-08-05","arxiv_id":"2308.04455","repositories_listed":2,"syntology":null},{"url":"/paper/ernie-sat-speech-and-text-joint-pretraining-1","title":"ERNIE-SAT: Speech and Text Joint Pretraining for Cross-Lingual Multi-Speaker Text-to-Speech","date":"2022-11-07","arxiv_id":"2211.03545","repositories_listed":2,"syntology":null},{"url":"/paper/neural-voice-cloning-with-a-few-samples","title":"Neural Voice Cloning with a Few Samples","date":"2018-02-14","arxiv_id":"1802.06006","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/few-shot-speech-deepfake-detection-adaptation","title":"Few-Shot Speech Deepfake Detection Adaptation with Gaussian Processes","date":"2025-05-29","arxiv_id":"2505.23619","repositories_listed":1,"syntology":null},{"url":"/paper/cloneval-an-open-voice-cloning-benchmark","title":"ClonEval: An Open Voice Cloning Benchmark","date":"2025-04-29","arxiv_id":"2504.20581","repositories_listed":1,"syntology":null},{"url":"/paper/speechdialoguefactory-generating-high-quality","title":"SpeechDialogueFactory: Generating High-Quality Speech Dialogue Data to Accelerate Your Speech-LLM Development","date":"2025-03-31","arxiv_id":"2503.23848","repositories_listed":1,"syntology":null},{"url":"/paper/2503-01710","title":"Spark-TTS: An Efficient LLM-Based Text-to-Speech Model with Single-Stream Decoupled Speech Tokens","date":"2025-03-03","arxiv_id":"2503.01710","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/songgen-a-single-stage-auto-regressive","title":"SongGen: A Single Stage Auto-regressive Transformer for Text-to-Song Generation","date":"2025-02-18","arxiv_id":"2502.13128","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/step-audio-unified-understanding-and","title":"Step-Audio: Unified Understanding and Generation in Intelligent Speech Interaction","date":"2025-02-17","arxiv_id":"2502.11946","repositories_listed":1,"syntology":null},{"url":"/paper/indextts-an-industrial-level-controllable-and","title":"IndexTTS: An Industrial-Level Controllable and Efficient Zero-Shot Text-To-Speech System","date":"2025-02-08","arxiv_id":"2502.05512","repositories_listed":1,"syntology":null},{"url":"/paper/lina-speech-gated-linear-attention-is-a-fast","title":"Lina-Speech: Gated Linear Attention is a Fast and Parameter-Efficient Learner for text-to-speech synthesis","date":"2024-10-30","arxiv_id":"2410.23320","repositories_listed":1,"syntology":null},{"url":"/paper/emoknob-enhance-voice-cloning-with-fine","title":"EmoKnob: Enhance Voice Cloning with Fine-Grained Emotion Control","date":"2024-10-01","arxiv_id":"2410.00316","repositories_listed":1,"syntology":null},{"url":"/paper/llamapartialspoof-an-llm-driven-fake-speech","title":"LlamaPartialSpoof: An LLM-Driven Fake Speech Dataset Simulating Disinformation Generation","date":"2024-09-23","arxiv_id":"2409.14743","repositories_listed":1,"syntology":null},{"url":"/paper/is-audio-spoof-detection-robust-to-laundering","title":"Is Audio Spoof Detection Robust to Laundering Attacks?","date":"2024-08-27","arxiv_id":"2408.14712","repositories_listed":1,"syntology":null},{"url":"/paper/cosyvoice-a-scalable-multilingual-zero-shot","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","date":"2024-07-07","arxiv_id":"2407.05407","repositories_listed":1,"syntology":null},{"url":"/paper/xtts-a-massively-multilingual-zero-shot-text","title":"XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model","date":"2024-06-07","arxiv_id":"2406.04904","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/small-e-small-language-model-with-linear","title":"Small-E: Small Language Model with Linear Attention for Efficient Speech Synthesis","date":"2024-06-06","arxiv_id":"2406.04467","repositories_listed":1,"syntology":null},{"url":"/paper/polyglotfake-a-novel-multilingual-and","title":"PolyGlotFake: A Novel Multilingual and Multimodal DeepFake Dataset","date":"2024-05-14","arxiv_id":"2405.08838","repositories_listed":1,"syntology":null},{"url":"/paper/styledubber-towards-multi-scale-style","title":"StyleDubber: Towards Multi-Scale Style Learning for Movie Dubbing","date":"2024-02-20","arxiv_id":"2402.12636","repositories_listed":1,"syntology":null},{"url":"/paper/proactive-detection-of-voice-cloning-with","title":"Proactive Detection of Voice Cloning with Localized Watermarking","date":"2024-01-30","arxiv_id":"2401.17264","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/openvoice-versatile-instant-voice-cloning","title":"OpenVoice: Versatile Instant Voice Cloning","date":"2023-12-03","arxiv_id":"2312.01479","repositories_listed":1,"syntology":{"n":15,"n_ran":3,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/single-and-multi-speaker-cloned-voice","title":"Single and Multi-Speaker Cloned Voice Detection: From Perceptual to Learned Features","date":"2023-07-15","arxiv_id":"2307.07683","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/low-resource-multilingual-and-zero-shot","title":"Low-Resource Multilingual and Zero-Shot Multispeaker TTS","date":"2022-10-21","arxiv_id":"2210.12223","repositories_listed":1,"syntology":null},{"url":"/paper/empirical-study-incorporating-linguistic","title":"Empirical Study Incorporating Linguistic Knowledge on Filled Pauses for Personalized Spontaneous Speech Synthesis","date":"2022-10-14","arxiv_id":"2210.07559","repositories_listed":1,"syntology":null},{"url":"/paper/dictionary-attacks-on-speaker-verification","title":"Dictionary Attacks on Speaker Verification","date":"2022-04-24","arxiv_id":"2204.11304","repositories_listed":1,"syntology":null}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}