{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/audio-driven-emotional-video-portraits","title":"Audio-Driven Emotional Video Portraits","arxiv_id":"2104.07452","date":"2021-04-15","proceeding":"CVPR 2021 1","authors":["Xinya Ji","Hang Zhou","Kaisiyuan Wang","Wayne Wu","Chen Change Loy","Xun Cao","Feng Xu"],"abstract":"Despite previous success in generating audio-driven talking heads, most of the previous studies focus on the correlation between speech content and the mouth shape. Facial emotion, which is one of the most important features on natural human faces, is always neglected in their methods. In this work, we present Emotional Video Portraits (EVP), a system for synthesizing high-quality video portraits with vivid emotional dynamics driven by audios. Specifically, we propose the Cross-Reconstructed Emotion Disentanglement technique to decompose speech into two decoupled spaces, i.e., a duration-independent emotion space and a duration dependent content space. With the disentangled features, dynamic 2D emotional facial landmarks can be deduced. Then we propose the Target-Adaptive Face Synthesis technique to generate the final high-quality video portraits, by bridging the gap between the deduced landmarks and the natural head poses of target videos. Extensive experiments demonstrate the effectiveness of our method both qualitatively and quantitatively.","url_abs":"https://arxiv.org/abs/2104.07452v2","url_pdf":"https://arxiv.org/pdf/2104.07452v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"audio-driven-emotional-video-portraits","repo_url":"https://github.com/jixinya/EVP","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"disentanglement","task_name":"Disentanglement"},{"task_slug":"face-generation","task_name":"Face Generation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2104.07452","atlas_url":"https://app.syntology.ai/?focus=2104.07452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.07452"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/jixinya/EVP","reach":null}],"summary":{"ran":5,"unverified":5},"by_repo_kind":{"listed":{"samples":10,"ran":5,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":10,"samples":[{"code_sha256_prefix":"41b35560a5a81913","entry":"Conv2dBlock","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"41b35560a5a81913"}},{"code_sha256_prefix":"eb7999f01975e8d3","entry":"DisBlock","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"eb7999f01975e8d3"}},{"code_sha256_prefix":"fbff1a137e8aad34","entry":"Discriminator","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"fbff1a137e8aad34"}},{"code_sha256_prefix":"688abc2ba2985313","entry":"DownSample","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"688abc2ba2985313"}},{"code_sha256_prefix":"8fdfb9ae9bda63a9","entry":"ToDisBlock","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8fdfb9ae9bda63a9"}},{"code_sha256_prefix":"746ff0b319646475","entry":"AutoEncoder2x","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"746ff0b319646475"}},{"code_sha256_prefix":"a895646bfeba77dc","entry":"Classify","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a895646bfeba77dc"}},{"code_sha256_prefix":"35a85937e6b3df78","entry":"Ct_encoder","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"35a85937e6b3df78"}},{"code_sha256_prefix":"28ff9409d185217b","entry":"Decoder","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"28ff9409d185217b"}},{"code_sha256_prefix":"8f760bf41e1a752f","entry":"EmotionNet","repo":"jixinya/EVP","repo_kind":"listed","path":"train/disentanglement/code/models_GAN.py","file_url":"https://github.com/jixinya/EVP/blob/HEAD/train/disentanglement/code/models_GAN.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"8f760bf41e1a752f"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}