{"url":"/task/talking-face-generation","name":"Talking Face Generation","slug":"talking-face-generation","description_markdown":"Talking face generation aims to synthesize a sequence of face images that correspond to given speech semantics\r\n\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Talking Face Generation by Adversarially Disentangled Audio-Visual Representation](https://github.com/Hangz-nju-cuhk/Talking-Face-Generation-DAVS) )</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":110,"papers_with_code":43,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":2,"parent_tasks":3},"benchmarks":[{"leaderboard":"/sota/talking-face-generation-on-crema-d","slug":"talking-face-generation-on-crema-d","dataset":"CREMA-D","dataset_url":"/dataset/crema-d","rows_in_archive":1,"metrics":["EmoAcc","FID","LSE-C"],"first_row_in_archive_order":{"model":"EmoGen","paper_title":"Emotionally Enhanced Talking Face Generation","paper_url":"/paper/emotionally-enhanced-talking-face-generation","paper_date":"2023-03-21","arxiv_id":"2303.11548","code_links":[{"title":"sahilg06/EmoGen","url":"https://github.com/sahilg06/EmoGen"}],"syntology":null}},{"leaderboard":"/sota/talking-face-generation-on-lrw","slug":"talking-face-generation-on-lrw","dataset":"LRW","dataset_url":"/dataset/lrw","rows_in_archive":1,"metrics":["LMD","SSIM"],"first_row_in_archive_order":{"model":"LipGAN","paper_title":"Towards Automatic Face-to-Face Translation","paper_url":"/paper/towards-automatic-face-to-face-translation-1","paper_date":"2020-03-01","arxiv_id":"2003.00418","code_links":[{"title":"Rudrabha/LipGAN","url":"https://github.com/Rudrabha/LipGAN"}],"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/pascal-voc","name":"PASCAL VOC","full_name":"PASCAL Visual Object Classes Challenge","num_papers_in_archive":198},{"url":"/dataset/lrw","name":"LRW","full_name":"Lip Reading in the Wild","num_papers_in_archive":188},{"url":"/dataset/vocaset","name":"VOCASET","full_name":"VOCASET","num_papers_in_archive":60},{"url":"/dataset/crema-d","name":"CREMA-D","full_name":"CREMA-D","num_papers_in_archive":28},{"url":"/dataset/glips","name":"GLips","full_name":"German Lips","num_papers_in_archive":6},{"url":"/dataset/animeceleb","name":"AnimeCeleb","full_name":"AnimeCeleb","num_papers_in_archive":2}],"subtasks":[{"url":"/task/constrained-lip-synchronization","name":"Constrained Lip-synchronization"},{"url":"/task/face-dubbing","name":"Face  Dubbing"}],"parent_tasks":[{"url":"/task/1-image-2-2-stitchi","name":"1 Image, 2*2 Stitchi"},{"url":"/task/10-shot-image-generation","name":"10-shot image generation"},{"url":"/task/face-generation","name":"Face Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":43,"tagged_in_all":110,"items":[{"url":"/paper/a-lip-sync-expert-is-all-you-need-for-speech","title":"A Lip Sync Expert Is All You Need for Speech to Lip Generation In The Wild","date":"2020-08-23","arxiv_id":"2008.10010","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/makeittalk-speaker-aware-talking-head","title":"MakeItTalk: Speaker-Aware Talking-Head Animation","date":"2020-04-27","arxiv_id":"2004.12992","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/real-time-neural-radiance-talking-portrait","title":"Real-time Neural Radiance Talking Portrait Synthesis via Audio-spatial Decomposition","date":"2022-11-22","arxiv_id":"2211.12368","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/joygen-audio-driven-3d-depth-aware-talking","title":"JoyGen: Audio-Driven 3D Depth-Aware Talking-Face Video Editing","date":"2025-01-03","arxiv_id":"2501.01798","repositories_listed":1,"syntology":null},{"url":"/paper/joint-co-speech-gesture-and-expressive","title":"Joint Co-Speech Gesture and Expressive Talking Face Generation using Diffusion with Adapters","date":"2024-12-18","arxiv_id":"2412.14333","repositories_listed":1,"syntology":null},{"url":"/paper/kan-based-fusion-of-dual-domain-for-audio","title":"KAN-Based Fusion of Dual-Domain for Audio-Driven Facial Landmarks Generation","date":"2024-09-09","arxiv_id":"2409.05330","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-talking-face-generation-by","title":"Controllable Talking Face Generation by Implicit Facial Keypoints Editing","date":"2024-06-05","arxiv_id":"2406.02880","repositories_listed":1,"syntology":null},{"url":"/paper/deepfake-generation-and-detection-a-benchmark","title":"Deepfake Generation and Detection: A Benchmark and Survey","date":"2024-03-26","arxiv_id":"2403.17881","repositories_listed":1,"syntology":null},{"url":"/paper/real3d-portrait-one-shot-realistic-3d-talking","title":"Real3D-Portrait: One-shot Realistic 3D Talking Portrait Synthesis","date":"2024-01-16","arxiv_id":"2401.08503","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/gsmoothface-generalized-smooth-talking-face","title":"GSmoothFace: Generalized Smooth Talking Face Generation via Fine Grained 3D Face Guidance","date":"2023-12-12","arxiv_id":"2312.07385","repositories_listed":1,"syntology":null},{"url":"/paper/neural-text-to-articulate-talk-deep-text-to","title":"Neural Text to Articulate Talk: Deep Text to Audiovisual Speech Synthesis achieving both Auditory and Photo-realism","date":"2023-12-11","arxiv_id":"2312.06613","repositories_listed":1,"syntology":null},{"url":"/paper/synctalk-the-devil-is-in-the-synchronization","title":"SyncTalk: The Devil is in the Synchronization for Talking Head Synthesis","date":"2023-11-29","arxiv_id":"2311.17590","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/diffdub-person-generic-visual-dubbing-using","title":"DiffDub: Person-generic Visual Dubbing Using Inpainting Renderer with Diffusion Auto-encoder","date":"2023-11-03","arxiv_id":"2311.01811","repositories_listed":1,"syntology":null},{"url":"/paper/hyperlips-hyper-control-lips-with-high","title":"HyperLips: Hyper Control Lips with High Resolution Decoder for Talking Face Generation","date":"2023-10-09","arxiv_id":"2310.05720","repositories_listed":1,"syntology":null},{"url":"/paper/hdtr-net-a-real-time-high-definition-teeth","title":"HDTR-Net: A Real-Time High-Definition Teeth Restoration Network for Arbitrary Talking Face Generation Methods","date":"2023-09-14","arxiv_id":"2309.07495","repositories_listed":1,"syntology":null},{"url":"/paper/identity-preserving-talking-face-generation","title":"Identity-Preserving Talking Face Generation with Landmark and Appearance Priors","date":"2023-05-15","arxiv_id":"2305.08293","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/seeing-what-you-said-talking-face-generation","title":"Seeing What You Said: Talking Face Generation Guided by a Lip Reading Expert","date":"2023-03-29","arxiv_id":"2303.17480","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/emotionally-enhanced-talking-face-generation","title":"Emotionally Enhanced Talking Face Generation","date":"2023-03-21","arxiv_id":"2303.11548","repositories_listed":1,"syntology":null},{"url":"/paper/dinet-deformation-inpainting-network-for","title":"DINet: Deformation Inpainting Network for Realistic Face Visually Dubbing on High Resolution Video","date":"2023-03-07","arxiv_id":"2303.03988","repositories_listed":1,"syntology":{"n":10,"n_ran":2,"n_unverified":8,"n_pointer_only":10}},{"url":"/paper/geneface-generalized-and-high-fidelity-audio","title":"GeneFace: Generalized and High-Fidelity Audio-Driven 3D Talking Face Synthesis","date":"2023-01-31","arxiv_id":"2301.13430","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/dpe-disentanglement-of-pose-and-expression","title":"DPE: Disentanglement of Pose and Expression for General Video Portrait Editing","date":"2023-01-16","arxiv_id":"2301.06281","repositories_listed":1,"syntology":{"n":23,"n_ran":15,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/styletalk-one-shot-talking-head-generation","title":"StyleTalk: One-shot Talking Head Generation with Controllable Speaking Styles","date":"2023-01-03","arxiv_id":"2301.01081","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/fnevr-neural-volume-rendering-for-face","title":"FNeVR: Neural Volume Rendering for Face Animation","date":"2022-09-21","arxiv_id":"2209.10340","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/learning-dynamic-facial-radiance-fields-for","title":"Learning Dynamic Facial Radiance Fields for Few-Shot Talking Head Synthesis","date":"2022-07-24","arxiv_id":"2207.11770","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/merkel-podcast-corpus-a-multimodal-dataset-1","title":"Merkel Podcast Corpus: A Multimodal Dataset Compiled from 16 Years of Angela Merkel’s Weekly Video Podcasts","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/merkel-podcast-corpus-a-multimodal-dataset","title":"Merkel Podcast Corpus: A Multimodal Dataset Compiled from 16 Years of Angela Merkel's Weekly Video Podcasts","date":"2022-05-24","arxiv_id":"2205.12194","repositories_listed":1,"syntology":null},{"url":"/paper/styleheat-one-shot-high-resolution-editable","title":"StyleHEAT: One-Shot High-Resolution Editable Talking Face Generation via Pre-trained StyleGAN","date":"2022-03-08","arxiv_id":"2203.04036","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/live-speech-portraits-real-time","title":"Live Speech Portraits: Real-Time Photorealistic Talking-Head Animation","date":"2021-09-22","arxiv_id":"2109.10595","repositories_listed":1,"syntology":null},{"url":"/paper/facial-synthesizing-dynamic-talking-face-with","title":"FACIAL: Synthesizing Dynamic Talking Face with Implicit Attribute Learning","date":"2021-08-18","arxiv_id":"2108.07938","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/high-speed-and-high-quality-text-to-lip","title":"Parallel and High-Fidelity Text-to-Lip Generation","date":"2021-07-14","arxiv_id":"2107.06831","repositories_listed":1,"syntology":null}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}