{"url":"/task/talking-head-generation","name":"Talking Head Generation","slug":"talking-head-generation","description_markdown":"Talking head generation is the task of generating a talking face from a set of images of a person.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Few-Shot Adversarial Learning of Realistic Neural Talking Head Models](https://arxiv.org/pdf/1905.08233v2.pdf) )</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":119,"papers_with_code":51,"benchmarks":7,"benchmark_tables_in_archive":7,"benchmark_tables_shown":7,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":1,"parent_tasks":2},"benchmarks":[{"leaderboard":"/sota/talking-head-generation-on-voxceleb2-1-shot","slug":"talking-head-generation-on-voxceleb2-1-shot","dataset":"VoxCeleb2 - 1-shot learning","dataset_url":"/dataset/voxceleb2","rows_in_archive":5,"metrics":["CSIM","LPIPS","Normalized Pose Error","SSIM","inference time (ms)","FID"],"first_row_in_archive_order":{"model":"Fast Bi-layer Avatars (medium size)","paper_title":"Fast Bi-layer Neural Synthesis of One-Shot Realistic Head Avatars","paper_url":"/paper/fast-bi-layer-neural-synthesis-of-one-shot","paper_date":"2020-08-24","arxiv_id":"2008.10174","code_links":[{"title":"saic-violet/bilayer-model","url":"https://github.com/saic-violet/bilayer-model"}],"syntology":null}},{"leaderboard":"/sota/talking-head-generation-on-voxceleb1-1-shot","slug":"talking-head-generation-on-voxceleb1-1-shot","dataset":"VoxCeleb1 - 1-shot learning","dataset_url":"/dataset/voxceleb1","rows_in_archive":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"Few-shot Adversarial Model","paper_title":"Few-Shot Adversarial Learning of Realistic Neural Talking Head Models","paper_url":"/paper/few-shot-adversarial-learning-of-realistic","paper_date":"2019-05-20","arxiv_id":"1905.08233","code_links":[{"title":"vincent-thevenin/Realistic-Neural-Talking-Head-Models","url":"https://github.com/vincent-thevenin/Realistic-Neural-Talking-Head-Models"},{"title":"grey-eye/talking-heads","url":"https://github.com/grey-eye/talking-heads"},{"title":"shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap","url":"https://github.com/shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap"},{"title":"ZVK/Talking-Heads","url":"https://github.com/ZVK/Talking-Heads"},{"title":"ZVK/talking_heads","url":"https://github.com/ZVK/talking_heads"},{"title":"Ierezell/PapierFewShot","url":"https://github.com/Ierezell/PapierFewShot"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/talking-head-generation-on-voxceleb1-32-shot","slug":"talking-head-generation-on-voxceleb1-32-shot","dataset":"VoxCeleb1 - 32-shot learning","dataset_url":"/dataset/voxceleb1","rows_in_archive":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"Few-shot Adversarial Model","paper_title":"Few-Shot Adversarial Learning of Realistic Neural Talking Head Models","paper_url":"/paper/few-shot-adversarial-learning-of-realistic","paper_date":"2019-05-20","arxiv_id":"1905.08233","code_links":[{"title":"vincent-thevenin/Realistic-Neural-Talking-Head-Models","url":"https://github.com/vincent-thevenin/Realistic-Neural-Talking-Head-Models"},{"title":"grey-eye/talking-heads","url":"https://github.com/grey-eye/talking-heads"},{"title":"shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap","url":"https://github.com/shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap"},{"title":"ZVK/Talking-Heads","url":"https://github.com/ZVK/Talking-Heads"},{"title":"ZVK/talking_heads","url":"https://github.com/ZVK/talking_heads"},{"title":"Ierezell/PapierFewShot","url":"https://github.com/Ierezell/PapierFewShot"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/talking-head-generation-on-voxceleb1-8-shot","slug":"talking-head-generation-on-voxceleb1-8-shot","dataset":"VoxCeleb1 - 8-shot learning","dataset_url":"/dataset/voxceleb1","rows_in_archive":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"Few-shot Adversarial Model","paper_title":"Few-Shot Adversarial Learning of Realistic Neural Talking Head Models","paper_url":"/paper/few-shot-adversarial-learning-of-realistic","paper_date":"2019-05-20","arxiv_id":"1905.08233","code_links":[{"title":"vincent-thevenin/Realistic-Neural-Talking-Head-Models","url":"https://github.com/vincent-thevenin/Realistic-Neural-Talking-Head-Models"},{"title":"grey-eye/talking-heads","url":"https://github.com/grey-eye/talking-heads"},{"title":"shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap","url":"https://github.com/shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap"},{"title":"ZVK/Talking-Heads","url":"https://github.com/ZVK/Talking-Heads"},{"title":"ZVK/talking_heads","url":"https://github.com/ZVK/talking_heads"},{"title":"Ierezell/PapierFewShot","url":"https://github.com/Ierezell/PapierFewShot"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}},{"leaderboard":"/sota/talking-head-generation-on-voxceleb2-8-shot","slug":"talking-head-generation-on-voxceleb2-8-shot","dataset":"VoxCeleb2 - 8-shot learning","dataset_url":"/dataset/voxceleb2","rows_in_archive":2,"metrics":["FID"],"first_row_in_archive_order":{"model":"CainGAN","paper_title":"Pose Manipulation with Identity Preservation","paper_url":"/paper/pose-manipulation-with-identity-preservation-1","paper_date":"2020-04-20","arxiv_id":"2004.09169","code_links":[],"syntology":null}},{"leaderboard":"/sota/talking-head-generation-on-100-sleep-nights","slug":"talking-head-generation-on-100-sleep-nights","dataset":"100 sleep nights of 8 caregivers","dataset_url":null,"rows_in_archive":1,"metrics":["10%"],"first_row_in_archive_order":{"model":"Ashok","paper_title":"0-Step Capturability, Motion Decomposition and Global Feedback Control of the 3D Variable Height-Inverted Pendulum","paper_url":"/paper/0-step-capturability-motion-decomposition-and","paper_date":"2019-12-12","arxiv_id":"1912.06078","code_links":[{"title":"GabrielEGC/IHMC-Robotics","url":"https://github.com/GabrielEGC/IHMC-Robotics"}],"syntology":null}},{"leaderboard":"/sota/talking-head-generation-on-voxceleb2-32-shot","slug":"talking-head-generation-on-voxceleb2-32-shot","dataset":"VoxCeleb2 - 32-shot learning","dataset_url":"/dataset/voxceleb2","rows_in_archive":1,"metrics":["FID"],"first_row_in_archive_order":{"model":"Few-shot Adversarial Model","paper_title":"Few-Shot Adversarial Learning of Realistic Neural Talking Head Models","paper_url":"/paper/few-shot-adversarial-learning-of-realistic","paper_date":"2019-05-20","arxiv_id":"1905.08233","code_links":[{"title":"vincent-thevenin/Realistic-Neural-Talking-Head-Models","url":"https://github.com/vincent-thevenin/Realistic-Neural-Talking-Head-Models"},{"title":"grey-eye/talking-heads","url":"https://github.com/grey-eye/talking-heads"},{"title":"shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap","url":"https://github.com/shoutOutYangJie/Few-Shot-Adversarial-Learning-for-face-swap"},{"title":"ZVK/Talking-Heads","url":"https://github.com/ZVK/Talking-Heads"},{"title":"ZVK/talking_heads","url":"https://github.com/ZVK/talking_heads"},{"title":"Ierezell/PapierFewShot","url":"https://github.com/Ierezell/PapierFewShot"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}}}],"datasets":[{"url":"/dataset/voxceleb1","name":"VoxCeleb1","full_name":"VoxCeleb1","num_papers_in_archive":680},{"url":"/dataset/voxceleb2","name":"VoxCeleb2","full_name":"VoxCeleb2","num_papers_in_archive":564},{"url":"/dataset/animeceleb","name":"AnimeCeleb","full_name":"AnimeCeleb","num_papers_in_archive":2}],"subtasks":[{"url":"/task/lip-sync","name":"Unconstrained Lip-synchronization"}],"parent_tasks":[{"url":"/task/10-shot-image-generation","name":"10-shot image generation"},{"url":"/task/face-generation","name":"Face Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":51,"tagged_in_all":119,"items":[{"url":"/paper/few-shot-adversarial-learning-of-realistic","title":"Few-Shot Adversarial Learning of Realistic Neural Talking Head Models","date":"2019-05-20","arxiv_id":"1905.08233","repositories_listed":6,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/a-lip-sync-expert-is-all-you-need-for-speech","title":"A Lip Sync Expert Is All You Need for Speech to Lip Generation In The Wild","date":"2020-08-23","arxiv_id":"2008.10010","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/makeittalk-speaker-aware-talking-head","title":"MakeItTalk: Speaker-Aware Talking-Head Animation","date":"2020-04-27","arxiv_id":"2004.12992","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/sadtalker-learning-realistic-3d-motion","title":"SadTalker: Learning Realistic 3D Motion Coefficients for Stylized Audio-Driven Single Image Talking Face Animation","date":"2022-11-22","arxiv_id":"2211.12194","repositories_listed":2,"syntology":null},{"url":"/paper/advancing-talking-head-generation-a","title":"Advancing Talking Head Generation: A Comprehensive Survey of Multi-Modal Methodologies, Datasets, Evaluation Metrics, and Loss Functions","date":"2025-06-23","arxiv_id":"2507.02900","repositories_listed":1,"syntology":null},{"url":"/paper/silence-is-golden-leveraging-adversarial","title":"Silence is Golden: Leveraging Adversarial Examples to Nullify Audio Control in LDM-based Talking-Head Generation","date":"2025-06-02","arxiv_id":"2506.01591","repositories_listed":1,"syntology":null},{"url":"/paper/audio-visual-controlled-video-diffusion-with","title":"Audio-visual Controlled Video Diffusion with Masked Selective State Spaces Modeling for Natural Talking Head Generation","date":"2025-04-03","arxiv_id":"2504.02542","repositories_listed":1,"syntology":null},{"url":"/paper/instag-learning-personalized-3d-talking-head","title":"InsTaG: Learning Personalized 3D Talking Head from Few-Second Video","date":"2025-02-27","arxiv_id":"2502.20387","repositories_listed":1,"syntology":null},{"url":"/paper/joint-co-speech-gesture-and-expressive","title":"Joint Co-Speech Gesture and Expressive Talking Face Generation using Diffusion with Adapters","date":"2024-12-18","arxiv_id":"2412.14333","repositories_listed":1,"syntology":null},{"url":"/paper/gohd-gaze-oriented-and-highly-disentangled","title":"GoHD: Gaze-oriented and Highly Disentangled Portrait Animation with Rhythmic Poses and Realistic Expression","date":"2024-12-12","arxiv_id":"2412.09296","repositories_listed":1,"syntology":null},{"url":"/paper/comparative-analysis-of-audio-feature","title":"Comparative Analysis of Audio Feature Extraction for Real-Time Talking Portrait Synthesis","date":"2024-11-20","arxiv_id":"2411.13209","repositories_listed":1,"syntology":null},{"url":"/paper/dawn-dynamic-frame-avatar-with-non","title":"DAWN: Dynamic Frame Avatar with Non-autoregressive Diffusion Framework for Talking Head Video Generation","date":"2024-10-17","arxiv_id":"2410.13726","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-fixed-topologies-unregistered-training","title":"Beyond Fixed Topologies: Unregistered Training and Comprehensive Evaluation Metrics for 3D Talking Heads","date":"2024-10-14","arxiv_id":"2410.11041","repositories_listed":1,"syntology":null},{"url":"/paper/emodiffhead-continuously-emotional-control-in","title":"EMOdiffhead: Continuously Emotional Control in Talking Head Generation via Diffusion","date":"2024-09-11","arxiv_id":"2409.07255","repositories_listed":1,"syntology":null},{"url":"/paper/moditalker-motion-disentangled-diffusion","title":"MoDiTalker: Motion-Disentangled Diffusion Model for High-Fidelity Talking Head Generation","date":"2024-03-28","arxiv_id":"2403.19144","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-super-resolution-for-one-shot","title":"Adaptive Super Resolution For One-Shot Talking-Head Generation","date":"2024-03-23","arxiv_id":"2403.15944","repositories_listed":1,"syntology":null},{"url":"/paper/emovoca-speech-driven-emotional-3d-talking","title":"EmoVOCA: Speech-Driven Emotional 3D Talking Heads","date":"2024-03-19","arxiv_id":"2403.12886","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-perceptual-quality","title":"A Comparative Study of Perceptual Quality Metrics for Audio-driven Talking Head Videos","date":"2024-03-11","arxiv_id":"2403.06421","repositories_listed":1,"syntology":null},{"url":"/paper/dreamtalk-when-expressive-talking-head","title":"DreamTalk: When Emotional Talking Head Generation Meets Diffusion Probabilistic Models","date":"2023-12-15","arxiv_id":"2312.09767","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/synctalk-the-devil-is-in-the-synchronization","title":"SyncTalk: The Devil is in the Synchronization for Talking Head Synthesis","date":"2023-11-29","arxiv_id":"2311.17590","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/diffdub-person-generic-visual-dubbing-using","title":"DiffDub: Person-generic Visual Dubbing Using Inpainting Renderer with Diffusion Auto-encoder","date":"2023-11-03","arxiv_id":"2311.01811","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-emotional-adaptation-for-audio","title":"Efficient Emotional Adaptation for Audio-Driven Talking-Head Generation","date":"2023-09-10","arxiv_id":"2309.04946","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/text-to-video-a-two-stage-framework-for-zero","title":"Text-to-Video: a Two-stage Framework for Zero-shot Identity-agnostic Talking-head Generation","date":"2023-08-12","arxiv_id":"2308.06457","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-identity-representation-conditioned","title":"Implicit Identity Representation Conditioned Memory Compensation Network for Talking Head video Generation","date":"2023-07-19","arxiv_id":"2307.09906","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/a-comprehensive-multi-scale-approach-for","title":"A Comprehensive Multi-scale Approach for Speech and Dynamics Synchrony in Talking Head Generation","date":"2023-07-04","arxiv_id":"2307.03270","repositories_listed":1,"syntology":null},{"url":"/paper/learning-landmarks-motion-from-speech-for","title":"Learning Landmarks Motion from Speech for Speaker-Agnostic 3D Talking Heads Generation","date":"2023-06-02","arxiv_id":"2306.01415","repositories_listed":1,"syntology":null},{"url":"/paper/renderme-360-a-large-digital-asset-library-1","title":"RenderMe-360: A Large Digital Asset Library and Benchmarks Towards High-fidelity Head Avatars","date":"2023-05-22","arxiv_id":"2305.13353","repositories_listed":1,"syntology":null},{"url":"/paper/dagan-depth-aware-generative-adversarial","title":"DaGAN++: Depth-Aware Generative Adversarial Network for Talking Head Video Generation","date":"2023-05-10","arxiv_id":"2305.06225","repositories_listed":1,"syntology":null},{"url":"/paper/face-animation-with-an-attribute-guided","title":"Face Animation with an Attribute-Guided Diffusion Model","date":"2023-04-06","arxiv_id":"2304.03199","repositories_listed":1,"syntology":{"n":12,"n_ran":3,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/emotionally-enhanced-talking-face-generation","title":"Emotionally Enhanced Talking Face Generation","date":"2023-03-21","arxiv_id":"2303.11548","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}