{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/from-audio-to-photoreal-embodiment","title":"From Audio to Photoreal Embodiment: Synthesizing Humans in Conversations","arxiv_id":"2401.01885","date":"2024-01-03","proceeding":"CVPR 2024 1","authors":["Evonne Ng","Javier Romero","Timur Bagautdinov","Shaojie Bai","Trevor Darrell","Angjoo Kanazawa","Alexander Richard"],"abstract":"We present a framework for generating full-bodied photorealistic avatars that gesture according to the conversational dynamics of a dyadic interaction. Given speech audio, we output multiple possibilities of gestural motion for an individual, including face, body, and hands. The key behind our method is in combining the benefits of sample diversity from vector quantization with the high-frequency details obtained through diffusion to generate more dynamic, expressive motion. We visualize the generated motion using highly photorealistic avatars that can express crucial nuances in gestures (e.g. sneers and smirks). To facilitate this line of research, we introduce a first-of-its-kind multi-view conversational dataset that allows for photorealistic reconstruction. Experiments show our model generates appropriate and diverse gestures, outperforming both diffusion- and VQ-only methods. Furthermore, our perceptual evaluation highlights the importance of photorealism (vs. meshes) in accurately assessing subtle motion details in conversational gestures. Code and dataset available online.","url_abs":"https://arxiv.org/abs/2401.01885v1","url_pdf":"https://arxiv.org/pdf/2401.01885v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"from-audio-to-photoreal-embodiment","repo_url":"https://github.com/facebookresearch/audio2photoreal","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"NOASSERTION"}}],"tasks":[{"task_slug":"diversity","task_name":"Diversity"},{"task_slug":"quantization","task_name":"Quantization"}],"methods":[{"method_slug":"diffusion","method_name":"Diffusion"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2401.01885","atlas_url":"https://app.syntology.ai/?focus=2401.01885","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01885"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/facebookresearch/audio2photoreal","reach":{"status":"ok","spdx":"NOASSERTION"}}],"summary":{"ran_violates":2,"ran_draft_wrong":2,"ran_honours":3,"ran":7,"ran_fixture":1,"unverified":3},"by_repo_kind":{"official":{"samples":18,"ran":15,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":18,"samples":[{"code_sha256_prefix":"7c49b92babb27a7d","entry":"default","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"model/vqvae.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/model/vqvae.py","link_basis":"plan_row","language":"python","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"7c49b92babb27a7d"}},{"code_sha256_prefix":"cfd76fd0d89574a4","entry":"approx_standard_normal_cdf","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/losses.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/losses.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"cfd76fd0d89574a4"}},{"code_sha256_prefix":"2ab2316ac6fdd869","entry":"betas_for_alpha_bar","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/gaussian_diffusion.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/gaussian_diffusion.py","link_basis":"harvester_set","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"2ab2316ac6fdd869"}},{"code_sha256_prefix":"d84976cfde065645","entry":"collate_v2","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"data_loaders/tensors.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/data_loaders/tensors.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"d84976cfde065645"}},{"code_sha256_prefix":"cd33283d615fb3d7","entry":"discretized_gaussian_log_likelihood","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/losses.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/losses.py","link_basis":"harvester_set","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"cd33283d615fb3d7"}},{"code_sha256_prefix":"3596c18df070c4af","entry":"get_model_args","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"utils/model_util.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/utils/model_util.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"3596c18df070c4af"}},{"code_sha256_prefix":"97cb206ca0491baf","entry":"get_named_beta_schedule","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/gaussian_diffusion.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/gaussian_diffusion.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"97cb206ca0491baf"}},{"code_sha256_prefix":"e41367ad14ff58fd","entry":"get_param_groups_and_shapes","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/fp16_util.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/fp16_util.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e41367ad14ff58fd"}},{"code_sha256_prefix":"315803aeb6b9f726","entry":"get_person_num","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"utils/model_util.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/utils/model_util.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"315803aeb6b9f726"}},{"code_sha256_prefix":"c360b5ccff704a88","entry":"laplace_smoothing","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"model/vqvae.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/model/vqvae.py","link_basis":"plan_row","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c360b5ccff704a88"}},{"code_sha256_prefix":"c984209a2c66afbe","entry":"make_beta_schedule","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"model/utils.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/model/utils.py","link_basis":"harvester_set","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"c984209a2c66afbe"}},{"code_sha256_prefix":"e20dd5102da3b050","entry":"make_master_params","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/fp16_util.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/fp16_util.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"e20dd5102da3b050"}},{"code_sha256_prefix":"cf2798b666b231ca","entry":"normal_kl","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/losses.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/losses.py","link_basis":"harvester_set","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"cf2798b666b231ca"}},{"code_sha256_prefix":"3c6433dd421724e0","entry":"prob_mask_like","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"model/utils.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/model/utils.py","link_basis":"harvester_set","language":"python","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"3c6433dd421724e0"}},{"code_sha256_prefix":"64fff1e30802b815","entry":"unflatten_master_params","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"diffusion/fp16_util.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/diffusion/fp16_util.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"64fff1e30802b815"}},{"code_sha256_prefix":"0b9f005ecb737755","entry":"collate_tensors","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"data_loaders/tensors.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/data_loaders/tensors.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0b9f005ecb737755"}},{"code_sha256_prefix":"09c8479d9a5b3e06","entry":"extract","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"model/utils.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/model/utils.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"09c8479d9a5b3e06"}},{"code_sha256_prefix":"80c46a6e9bfa68f7","entry":"lengths_to_mask","repo":"facebookresearch/audio2photoreal","repo_kind":"official","path":"data_loaders/tensors.py","file_url":"https://github.com/facebookresearch/audio2photoreal/blob/HEAD/data_loaders/tensors.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"80c46a6e9bfa68f7"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}