{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/neural-voice-cloning-with-a-few-samples","title":"Neural Voice Cloning with a Few Samples","arxiv_id":"1802.06006","date":"2018-02-14","proceeding":"NeurIPS 2018 12","authors":["Sercan O. Arik","Jitong Chen","Kainan Peng","Wei Ping","Yanqi Zhou"],"abstract":"Voice cloning is a highly desired feature for personalized speech interfaces.\nNeural network based speech synthesis has been shown to generate high quality\nspeech for a large number of speakers. In this paper, we introduce a neural\nvoice cloning system that takes a few audio samples as input. We study two\napproaches: speaker adaptation and speaker encoding. Speaker adaptation is\nbased on fine-tuning a multi-speaker generative model with a few cloning\nsamples. Speaker encoding is based on training a separate model to directly\ninfer a new speaker embedding from cloning audios and to be used with a\nmulti-speaker generative model. In terms of naturalness of the speech and its\nsimilarity to original speaker, both approaches can achieve good performance,\neven with very few cloning audios. While speaker adaptation can achieve better\nnaturalness and similarity, the cloning time or required memory for the speaker\nencoding approach is significantly less, making it favorable for low-resource\ndeployment.","url_abs":"http://arxiv.org/abs/1802.06006v3","url_pdf":"http://arxiv.org/pdf/1802.06006v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"neural-voice-cloning-with-a-few-samples","repo_url":"https://github.com/SforAiDl/Neural-Voice-Cloning-With-Few-Samples","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"neural-voice-cloning-with-a-few-samples","repo_url":"https://github.com/jackaduma/CycleGAN-VC2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"speech-synthesis","task_name":"Speech Synthesis"},{"task_slug":"voice-cloning","task_name":"Voice Cloning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1802.06006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.06006"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jackaduma/CycleGAN-VC2","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/SforAiDl/Neural-Voice-Cloning-With-Few-Samples","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"unverified":2},"by_repo_kind":{"listed":{"samples":2,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"867f5af1413f6123","entry":"get_speaker_embeddings","repo":"SforAiDl/Neural-Voice-Cloning-With-Few-Samples","repo_kind":"listed","path":"train_whole.py","file_url":"https://github.com/SforAiDl/Neural-Voice-Cloning-With-Few-Samples/blob/HEAD/train_whole.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"867f5af1413f6123"}},{"code_sha256_prefix":"d1e8be9e1dc33f06","entry":"load_checkpoint","repo":"SforAiDl/Neural-Voice-Cloning-With-Few-Samples","repo_kind":"listed","path":"train_whole.py","file_url":"https://github.com/SforAiDl/Neural-Voice-Cloning-With-Few-Samples/blob/HEAD/train_whole.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"d1e8be9e1dc33f06"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}