{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/learning-latent-representations-for-style","title":"Learning latent representations for style control and transfer in end-to-end speech synthesis","arxiv_id":"1812.04342","date":"2018-12-11","proceeding":null,"authors":["Ya-Jie Zhang","Shifeng Pan","Lei He","Zhen-Hua Ling"],"abstract":"In this paper, we introduce the Variational Autoencoder (VAE) to an\nend-to-end speech synthesis model, to learn the latent representation of\nspeaking styles in an unsupervised manner. The style representation learned\nthrough VAE shows good properties such as disentangling, scaling, and\ncombination, which makes it easy for style control. Style transfer can be\nachieved in this framework by first inferring style representation through the\nrecognition network of VAE, then feeding it into TTS network to guide the style\nin synthesizing speech. To avoid Kullback-Leibler (KL) divergence collapse in\ntraining, several techniques are adopted. Finally, the proposed model shows\ngood performance of style control and outperforms Global Style Token (GST)\nmodel in ABX preference tests on style transfer.","url_abs":"http://arxiv.org/abs/1812.04342v2","url_pdf":"http://arxiv.org/pdf/1812.04342v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"learning-latent-representations-for-style","repo_url":"https://github.com/jinhan/tacotron2-vae","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"BSD-3-Clause"}},{"paper_slug":"learning-latent-representations-for-style","repo_url":"https://github.com/yanggeng1995/vae_tacotron","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"speech-synthesis","task_name":"Speech Synthesis"},{"task_slug":"style-transfer","task_name":"Style Transfer"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1812.04342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.04342"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/yanggeng1995/vae_tacotron","reach":{"status":"ok","spdx":"MIT"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jinhan/tacotron2-vae","reach":{"status":"ok","spdx":"BSD-3-Clause"}}],"summary":{"ran_draft_wrong":2,"unverified":6},"by_repo_kind":{"listed":{"samples":8,"ran":2,"repositories":2}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":7,"samples":[{"code_sha256_prefix":"83e2c862f7374ac7","entry":"dynamic_range_compression","repo":"jinhan/tacotron2-vae","repo_kind":"listed","path":"audio_processing.py","file_url":"https://github.com/jinhan/tacotron2-vae/blob/HEAD/audio_processing.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"83e2c862f7374ac7"}},{"code_sha256_prefix":"9f9ec26d4cdfdf7d","entry":"griffin_lim","repo":"jinhan/tacotron2-vae","repo_kind":"listed","path":"audio_processing.py","file_url":"https://github.com/jinhan/tacotron2-vae/blob/HEAD/audio_processing.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"9f9ec26d4cdfdf7d"}},{"code_sha256_prefix":"f7bf9f0265e33cf0","entry":"apply_gradient_allreduce","repo":"jinhan/tacotron2-vae","repo_kind":"listed","path":"distributed.py","file_url":"https://github.com/jinhan/tacotron2-vae/blob/HEAD/distributed.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"f7bf9f0265e33cf0"}},{"code_sha256_prefix":"f887d5ffcbc6ad3b","entry":"conversion_helper","repo":"jinhan/tacotron2-vae","repo_kind":"listed","path":"fp16_optimizer.py","file_url":"https://github.com/jinhan/tacotron2-vae/blob/HEAD/fp16_optimizer.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"f887d5ffcbc6ad3b"}},{"code_sha256_prefix":"f91fc9b0bec10110","entry":"create_hparams","repo":"jinhan/tacotron2-vae","repo_kind":"listed","path":"hparams.py","file_url":"https://github.com/jinhan/tacotron2-vae/blob/HEAD/hparams.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"f91fc9b0bec10110"}},{"code_sha256_prefix":"38034aa1d784464d","entry":"fp16_to_fp32","repo":"jinhan/tacotron2-vae","repo_kind":"listed","path":"fp16_optimizer.py","file_url":"https://github.com/jinhan/tacotron2-vae/blob/HEAD/fp16_optimizer.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"38034aa1d784464d"}},{"code_sha256_prefix":"ea22baceb9a219a1","entry":"fp32_to_fp16","repo":"jinhan/tacotron2-vae","repo_kind":"listed","path":"fp16_optimizer.py","file_url":"https://github.com/jinhan/tacotron2-vae/blob/HEAD/fp16_optimizer.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"mcp_get_code":{"code_sha256":"ea22baceb9a219a1"}},{"code_sha256_prefix":"8bb2d1c4dd2ef71e","entry":"prenet","repo":"yanggeng1995/vae_tacotron","repo_kind":"listed","path":"models/modules.py","file_url":"https://github.com/yanggeng1995/vae_tacotron/blob/HEAD/models/modules.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"8bb2d1c4dd2ef71e"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}