{"url":"/task/motion-synthesis","name":"Motion Synthesis","slug":"motion-synthesis","description_markdown":"Creating a video where people in the images move (such as blinking or smiling) requires specialized AI technology like Deepfake or Motion Synthesis, which I cannot do directly here.  \r\n\r\nHowever, if you want a video made from the images you provided, I can create an animated slideshow with cute effects and royalty-free background music. Would you be interested in this?","categories":[{"name":"Computer Code","url":"/area/computer-code"},{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":282,"papers_with_code":126,"benchmarks":13,"benchmark_tables_in_archive":13,"benchmark_tables_shown":13,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":19,"subtasks":3,"parent_tasks":2},"benchmarks":[{"leaderboard":"/sota/motion-synthesis-on-humanml3d","slug":"motion-synthesis-on-humanml3d","dataset":"HumanML3D","dataset_url":"/dataset/humanml3d","rows_in_archive":37,"metrics":["FID","R Precision Top3","Diversity","Multimodality"],"first_row_in_archive_order":{"model":"Motion Anything","paper_title":"Motion Anything: Any to Motion Generation","paper_url":"/paper/motion-anything-any-to-motion-generation","paper_date":"2025-03-10","arxiv_id":"2503.06955","code_links":[{"title":"steve-zeyu-zhang/MotionAnything","url":"https://github.com/steve-zeyu-zhang/MotionAnything"}],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-kit-motion-language","slug":"motion-synthesis-on-kit-motion-language","dataset":"KIT Motion-Language","dataset_url":"/dataset/kit-motion-language","rows_in_archive":31,"metrics":["FID","R Precision Top3","Diversity","Multimodality"],"first_row_in_archive_order":{"model":"Motion Anything","paper_title":"Motion Anything: Any to Motion Generation","paper_url":"/paper/motion-anything-any-to-motion-generation","paper_date":"2025-03-10","arxiv_id":"2503.06955","code_links":[{"title":"steve-zeyu-zhang/MotionAnything","url":"https://github.com/steve-zeyu-zhang/MotionAnything"}],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-aist","slug":"motion-synthesis-on-aist","dataset":"AIST++","dataset_url":"/dataset/aist","rows_in_archive":12,"metrics":["FID","Beat alignment score"],"first_row_in_archive_order":{"model":"Motion Anything","paper_title":"Motion Anything: Any to Motion Generation","paper_url":"/paper/motion-anything-any-to-motion-generation","paper_date":"2025-03-10","arxiv_id":"2503.06955","code_links":[{"title":"steve-zeyu-zhang/MotionAnything","url":"https://github.com/steve-zeyu-zhang/MotionAnything"}],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-interhuman","slug":"motion-synthesis-on-interhuman","dataset":"InterHuman","dataset_url":"/dataset/interhuman","rows_in_archive":10,"metrics":["FID","R-Precision Top3","MMDist","MModality"],"first_row_in_archive_order":{"model":"InterMask","paper_title":"InterMask: 3D Human Interaction Generation via Collaborative Masked Modelling","paper_url":"/paper/intermask-3d-human-interaction-generation-via","paper_date":"2024-10-13","arxiv_id":"2410.10010","code_links":[{"title":"gohar-malik/intermask","url":"https://github.com/gohar-malik/intermask"}],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-finedance","slug":"motion-synthesis-on-finedance","dataset":"FineDance","dataset_url":"/dataset/finedance","rows_in_archive":8,"metrics":["fid_k","BAS"],"first_row_in_archive_order":{"model":"DanceMosaic","paper_title":"DanceMosaic: High-Fidelity Dance Generation with Multimodal Editability","paper_url":"/paper/dancemosaic-high-fidelity-dance-generation","paper_date":"2025-04-06","arxiv_id":"2504.04634","code_links":[],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-inter-x","slug":"motion-synthesis-on-inter-x","dataset":"Inter-X","dataset_url":"/dataset/inter-x","rows_in_archive":6,"metrics":["FID","R-Precision Top3","MMDist","MModality"],"first_row_in_archive_order":{"model":"InterMask","paper_title":"InterMask: 3D Human Interaction Generation via Collaborative Masked Modelling","paper_url":"/paper/intermask-3d-human-interaction-generation-via","paper_date":"2024-10-13","arxiv_id":"2410.10010","code_links":[{"title":"gohar-malik/intermask","url":"https://github.com/gohar-malik/intermask"}],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-aioz-gdance","slug":"motion-synthesis-on-aioz-gdance","dataset":"AIOZ-GDANCE","dataset_url":"/dataset/aioz-gdance","rows_in_archive":4,"metrics":["FID","MMC","GenDiv","PFC","GMR","GMC","TIF"],"first_row_in_archive_order":{"model":"Scalable Group Choreography","paper_title":"Scalable Group Choreography via Variational Phase Manifold Learning","paper_url":"/paper/scalable-group-choreography-via-variational","paper_date":"2024-07-26","arxiv_id":"2407.18839","code_links":[],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-lafan1","slug":"motion-synthesis-on-lafan1","dataset":"LaFAN1","dataset_url":"/dataset/lafan1","rows_in_archive":4,"metrics":["L2Q@5","L2Q@15","L2Q@30","L2P@5","L2P@15","L2P@30","NPSS@5","NPSS@15","NPSS@30"],"first_row_in_archive_order":{"model":"$\\Delta$-interpolator","paper_title":"Motion Inbetweening via Deep $Δ$-Interpolator","paper_url":"/paper/motion-inbetweening-via-deep-d-interpolator","paper_date":"2022-01-18","arxiv_id":"2201.06701","code_links":[{"title":"boreshkinai/delta-interpolator","url":"https://github.com/boreshkinai/delta-interpolator"}],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-motion-x","slug":"motion-synthesis-on-motion-x","dataset":"Motion-X","dataset_url":"/dataset/motion-x","rows_in_archive":4,"metrics":["FID","TMR-R-Precision Top3","TMR-Matching Score","MModality","Diversity"],"first_row_in_archive_order":{"model":"HumanTOMATO","paper_title":"HumanTOMATO: Text-aligned Whole-body Motion Generation","paper_url":"/paper/humantomato-text-aligned-whole-body-motion","paper_date":"2023-10-19","arxiv_id":"2310.12978","code_links":[{"title":"IDEA-Research/HumanTOMATO","url":"https://github.com/IDEA-Research/HumanTOMATO"}],"syntology":{"n":18,"n_ran":12,"n_unverified":6,"n_pointer_only":18}}},{"leaderboard":"/sota/motion-synthesis-on-brace","slug":"motion-synthesis-on-brace","dataset":"BRACE","dataset_url":"/dataset/brace","rows_in_archive":3,"metrics":["Frechet Inception Distance","Beat alignment score","Beat DTW cost","Footwork average","Powermove average","Toprock average"],"first_row_in_archive_order":{"model":"Dance Revolution","paper_title":"Dance Revolution: Long-Term Dance Generation with Music via Curriculum Learning","paper_url":"/paper/dance-revolution-long-sequence-dance","paper_date":"2020-06-11","arxiv_id":"2006.06119","code_links":[],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-humanact12","slug":"motion-synthesis-on-humanact12","dataset":"HumanAct12","dataset_url":"/dataset/humanact12","rows_in_archive":2,"metrics":["Accuracy","FID","Multimodality"],"first_row_in_archive_order":{"model":"MDM","paper_title":"Human Motion Diffusion Model","paper_url":"/paper/human-motion-diffusion-model","paper_date":"2022-09-29","arxiv_id":"2209.14916","code_links":[{"title":"guytevet/motion-diffusion-model","url":"https://github.com/guytevet/motion-diffusion-model"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/motion-synthesis-on-tmd","slug":"motion-synthesis-on-tmd","dataset":"TMD","dataset_url":"/dataset/tmd","rows_in_archive":1,"metrics":["FID","BAS","MModality","MMDist"],"first_row_in_archive_order":{"model":"Motion Anything","paper_title":"Motion Anything: Any to Motion Generation","paper_url":"/paper/motion-anything-any-to-motion-generation","paper_date":"2025-03-10","arxiv_id":"2503.06955","code_links":[{"title":"steve-zeyu-zhang/MotionAnything","url":"https://github.com/steve-zeyu-zhang/MotionAnything"}],"syntology":null}},{"leaderboard":"/sota/motion-synthesis-on-trinity-speech-gesture","slug":"motion-synthesis-on-trinity-speech-gesture","dataset":"Trinity Speech-Gesture Dataset","dataset_url":"/dataset/https-trinityspeechgesture-scss-tcd-ie","rows_in_archive":1,"metrics":["Mean Opinion Score"],"first_row_in_archive_order":{"model":"Match-TTSG","paper_title":"Unified speech and gesture synthesis using flow matching","paper_url":"/paper/unified-speech-and-gesture-synthesis-using","paper_date":"2023-10-08","arxiv_id":"2310.05181","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/humanml3d","name":"HumanML3D","full_name":"","num_papers_in_archive":201},{"url":"/dataset/kit-motion-language","name":"KIT Motion-Language","full_name":"","num_papers_in_archive":48},{"url":"/dataset/humanact12","name":"HumanAct12","full_name":"","num_papers_in_archive":37},{"url":"/dataset/interhuman","name":"InterHuman","full_name":"","num_papers_in_archive":24},{"url":"/dataset/arctic","name":"ARCTIC","full_name":"Articulated Objects in Free-form Hand Interaction","num_papers_in_archive":23},{"url":"/dataset/motion-x","name":"Motion-X","full_name":"","num_papers_in_archive":22},{"url":"/dataset/aist","name":"AIST++","full_name":"","num_papers_in_archive":21},{"url":"/dataset/finedance","name":"FineDance","full_name":"","num_papers_in_archive":18},{"url":"/dataset/inter-x","name":"Inter-X","full_name":"","num_papers_in_archive":14},{"url":"/dataset/aioz-gdance","name":"AIOZ-GDANCE","full_name":"","num_papers_in_archive":7},{"url":"/dataset/lafan1","name":"LaFAN1","full_name":"Ubisoft La Forge Animation Dataset","num_papers_in_archive":7},{"url":"/dataset/brace","name":"BRACE","full_name":"The Breakdancing Competition Dataset for Dance Motion Synthesis","num_papers_in_archive":5},{"url":"/dataset/dd100","name":"DD100","full_name":"","num_papers_in_archive":5},{"url":"/dataset/chairs-dataset","name":"CHAIRS dataset","full_name":"","num_papers_in_archive":3},{"url":"/dataset/tmd","name":"TMD","full_name":"Text-Music-Dance","num_papers_in_archive":2},{"url":"/dataset/https-trinityspeechgesture-scss-tcd-ie","name":"Trinity Speech-Gesture Dataset","full_name":"","num_papers_in_archive":2},{"url":"/dataset/vrmocap-vr-mocap-dataset-for-pose","name":"VRMocap: VR Mocap Dataset for Pose Reconstruction","full_name":"","num_papers_in_archive":2},{"url":"/dataset/both57m","name":"BOTH57M","full_name":"","num_papers_in_archive":1},{"url":"/dataset/guitar-playing-motion-dataset","name":"Guitar Playing Motion Dataset","full_name":"Guitar Playing Motion Dataset","num_papers_in_archive":1}],"subtasks":[{"url":"/task/motion-in-betweening","name":"motion in-betweening"},{"url":"/task/motion-style-transfer","name":"Motion Style Transfer"},{"url":"/task/temporal-human-motion-composition","name":"Temporal Human Motion Composition"}],"parent_tasks":[{"url":"/task/10-shot-image-generation","name":"10-shot image generation"},{"url":"/task/3d-human-pose-tracking","name":"3D Human Pose Tracking"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":126,"tagged_in_all":282,"items":[{"url":"/paper/on-human-motion-prediction-using-recurrent","title":"On human motion prediction using recurrent neural networks","date":"2017-05-06","arxiv_id":"1705.02445","repositories_listed":8,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/deepmimic-example-guided-deep-reinforcement","title":"DeepMimic: Example-Guided Deep Reinforcement Learning of Physics-Based Character Skills","date":"2018-04-08","arxiv_id":"1804.02717","repositories_listed":6,"syntology":null},{"url":"/paper/multi-view-motion-synthesis-via-applying","title":"Multi-View Motion Synthesis via Applying Rotated Dual-Pixel Blur Kernels","date":"2021-11-15","arxiv_id":"2111.07837","repositories_listed":3,"syntology":null},{"url":"/paper/moglow-probabilistic-and-controllable-motion","title":"MoGlow: Probabilistic and controllable motion synthesis using normalising flows","date":"2019-05-16","arxiv_id":"1905.06598","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/hp-gan-probabilistic-3d-human-motion","title":"HP-GAN: Probabilistic 3D human motion prediction via GAN","date":"2017-11-27","arxiv_id":"1711.09561","repositories_listed":3,"syntology":null},{"url":"/paper/anymole-any-character-motion-in-betweening","title":"AnyMoLe: Any Character Motion In-betweening Leveraging Video Diffusion Models","date":"2025-03-11","arxiv_id":"2503.08417","repositories_listed":2,"syntology":null},{"url":"/paper/motionclone-training-free-motion-cloning-for","title":"MotionClone: Training-Free Motion Cloning for Controllable Video Generation","date":"2024-06-08","arxiv_id":"2406.05338","repositories_listed":2,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":15}},{"url":"/paper/human-motion-diffusion-as-a-generative-prior","title":"Human Motion Diffusion as a Generative Prior","date":"2023-03-02","arxiv_id":"2303.01418","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":2}},{"url":"/paper/motiondiffuse-text-driven-human-motion","title":"MotionDiffuse: Text-Driven Human Motion Generation with Diffusion Model","date":"2022-08-31","arxiv_id":"2208.15001","repositories_listed":2,"syntology":null},{"url":"/paper/meshtalk-3d-face-animation-from-speech-using","title":"MeshTalk: 3D Face Animation from Speech using Cross-Modality Disentanglement","date":"2021-04-16","arxiv_id":"2104.08223","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/action-conditioned-3d-human-motion-synthesis","title":"Action-Conditioned 3D Human Motion Synthesis with Transformer VAE","date":"2021-04-12","arxiv_id":"2104.05670","repositories_listed":2,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/dancing-to-music","title":"Dancing to Music","date":"2019-11-05","arxiv_id":"1911.02001","repositories_listed":2,"syntology":null},{"url":"/paper/deepgesture-a-conversational-gesture","title":"DeepGesture: A conversational gesture synthesis system based on emotions and semantics","date":"2025-07-03","arxiv_id":"2507.03147","repositories_listed":1,"syntology":null},{"url":"/paper/volumetricsmpl-a-neural-volumetric-body-model","title":"VolumetricSMPL: A Neural Volumetric Body Model for Efficient Interactions, Contacts, and Collisions","date":"2025-06-29","arxiv_id":"2506.23236","repositories_listed":1,"syntology":null},{"url":"/paper/duetgen-music-driven-two-person-dance","title":"DuetGen: Music Driven Two-Person Dance Generation via Hierarchical Masked Modeling","date":"2025-06-23","arxiv_id":"2506.18680","repositories_listed":1,"syntology":null},{"url":"/paper/generative-ai-for-character-animation-a","title":"Generative AI for Character Animation: A Comprehensive Survey of Techniques, Applications, and Future Directions","date":"2025-04-27","arxiv_id":"2504.19056","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-motion-imitation","title":"Reinforcement learning-based motion imitation for physiologically plausible musculoskeletal motor control","date":"2025-03-18","arxiv_id":"2503.14637","repositories_listed":1,"syntology":null},{"url":"/paper/motion-anything-any-to-motion-generation","title":"Motion Anything: Any to Motion Generation","date":"2025-03-10","arxiv_id":"2503.06955","repositories_listed":1,"syntology":null},{"url":"/paper/free-t2m-frequency-enhanced-text-to-motion","title":"Free-T2M: Frequency Enhanced Text-to-Motion Diffusion Model With Consistency Loss","date":"2025-01-30","arxiv_id":"2501.18232","repositories_listed":1,"syntology":null},{"url":"/paper/keta-kinematic-phrases-enhanced-text-to","title":"KETA: Kinematic-Phrases-Enhanced Text-to-Motion Generation via Fine-grained Alignment","date":"2025-01-25","arxiv_id":"2501.15058","repositories_listed":1,"syntology":null},{"url":"/paper/motionllama-a-unified-framework-for-motion","title":"MotionLLaMA: A Unified Framework for Motion Synthesis and Comprehension","date":"2024-11-26","arxiv_id":"2411.17335","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-motion-in-text-to-video-generation","title":"Enhancing Motion in Text-to-Video Generation with Decomposed Encoding and Conditioning","date":"2024-10-31","arxiv_id":"2410.24219","repositories_listed":1,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/sitcom-crafter-a-plot-driven-human-motion","title":"Sitcom-Crafter: A Plot-Driven Human Motion Generation System in 3D Scenes","date":"2024-10-14","arxiv_id":"2410.10790","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/intermask-3d-human-interaction-generation-via","title":"InterMask: 3D Human Interaction Generation via Collaborative Masked Modelling","date":"2024-10-13","arxiv_id":"2410.10010","repositories_listed":1,"syntology":null},{"url":"/paper/bad-bidirectional-auto-regressive-diffusion","title":"BAD: Bidirectional Auto-regressive Diffusion for Text-to-Motion Generation","date":"2024-09-17","arxiv_id":"2409.10847","repositories_listed":1,"syntology":null},{"url":"/paper/lawdnet-enhanced-audio-driven-lip-synthesis","title":"LawDNet: Enhanced Audio-Driven Lip Synthesis via Local Affine Warping Deformation","date":"2024-09-14","arxiv_id":"2409.09326","repositories_listed":1,"syntology":null},{"url":"/paper/t3m-text-guided-3d-human-motion-synthesis","title":"T3M: Text Guided 3D Human Motion Synthesis from Speech","date":"2024-08-23","arxiv_id":"2408.12885","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21136","title":"MotionCraft: Crafting Whole-Body Motion with Plug-and-Play Multimodal Controls","date":"2024-07-30","arxiv_id":"2407.21136","repositories_listed":1,"syntology":null},{"url":"/paper/length-aware-motion-synthesis-via-latent","title":"Length-Aware Motion Synthesis via Latent Diffusion","date":"2024-07-16","arxiv_id":"2407.11532","repositories_listed":1,"syntology":null},{"url":"/paper/monkey-see-monkey-do-harnessing-self","title":"Monkey See, Monkey Do: Harnessing Self-attention in Motion Diffusion for Zero-shot Motion Transfer","date":"2024-06-10","arxiv_id":"2406.06508","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}