{"url":"/task/gesture-generation","name":"Gesture Generation","slug":"gesture-generation","description_markdown":"Generation of gestures, as a sequence of 3d poses","categories":[{"name":"Robots","url":"/area/robots"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":107,"papers_with_code":46,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":7,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/gesture-generation-on-beat2","slug":"gesture-generation-on-beat2","dataset":"BEAT2","dataset_url":"/dataset/beat2","rows_in_archive":14,"metrics":["FGD"],"first_row_in_archive_order":{"model":"Intentional Gesture","paper_title":"Intentional Gesture: Deliver Your Intentions with Gestures for Speech","paper_url":"/paper/intentional-gesture-deliver-your-intentions","paper_date":"2025-05-21","arxiv_id":"2505.15197","code_links":[{"title":"andypinxinliu/Intentional-Gesture","url":"https://github.com/andypinxinliu/Intentional-Gesture"}],"syntology":null}},{"leaderboard":"/sota/gesture-generation-on-ted-gesture-dataset","slug":"gesture-generation-on-ted-gesture-dataset","dataset":"TED Gesture Dataset","dataset_url":"/dataset/ted-gesture-dataset","rows_in_archive":6,"metrics":["FGD"],"first_row_in_archive_order":{"model":"AQ-GT","paper_title":"AQ-GT: a Temporally Aligned and Quantized GRU-Transformer for Co-Speech Gesture Synthesis","paper_url":"/paper/aq-gt-a-temporally-aligned-and-quantized-gru","paper_date":"2023-05-02","arxiv_id":"2305.01241","code_links":[{"title":"hvoss-techfak/AQGT","url":"https://github.com/hvoss-techfak/AQGT"}],"syntology":null}},{"leaderboard":"/sota/gesture-generation-on-beat","slug":"gesture-generation-on-beat","dataset":"BEAT","dataset_url":"/dataset/beat","rows_in_archive":5,"metrics":["FID"],"first_row_in_archive_order":{"model":"CaMN","paper_title":"BEAT: A Large-Scale Semantic and Emotional Multi-Modal Dataset for Conversational Gestures Synthesis","paper_url":"/paper/beat-a-large-scale-semantic-and-emotional","paper_date":"2022-03-10","arxiv_id":"2203.05297","code_links":[{"title":"PantoMatrix/PantoMatrix","url":"https://github.com/PantoMatrix/PantoMatrix"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/gesture-generation-on-dvs128-gesture","slug":"gesture-generation-on-dvs128-gesture","dataset":"DVS128 Gesture","dataset_url":"/dataset/dvs128-gesture-dataset","rows_in_archive":1,"metrics":["Accuracy (% )"],"first_row_in_archive_order":{"model":"ECSNet","paper_title":"Ecsnet: Spatio-temporal feature learning for event camera","paper_url":"/paper/ecsnet-spatio-temporal-feature-learning-for","paper_date":"2022-08-29","arxiv_id":null,"code_links":[{"title":"happychenpipi/ECSNet","url":"https://github.com/happychenpipi/ECSNet"}],"syntology":null}}],"datasets":[{"url":"/dataset/dvs128-gesture-dataset","name":"DVS128 Gesture","full_name":"","num_papers_in_archive":103},{"url":"/dataset/beat","name":"BEAT","full_name":"Body-Expression-Audio-Text","num_papers_in_archive":56},{"url":"/dataset/beat2","name":"BEAT2","full_name":"BEAT-SMPLX-FLAME","num_papers_in_archive":23},{"url":"/dataset/ted-gesture-dataset","name":"TED Gesture Dataset","full_name":"","num_papers_in_archive":12},{"url":"/dataset/talking-with-hands-16-2m","name":"Talking With Hands 16.2M","full_name":"Talking With Hands 16.2M","num_papers_in_archive":5},{"url":"/dataset/bige","name":"BiGe","full_name":"Bielefeld Gesture Corpus","num_papers_in_archive":2},{"url":"/dataset/llmafia","name":"LLMafia","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/3d-shape-generation","name":"3D Shape Generation"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":46,"tagged_in_all":107,"items":[{"url":"/paper/robosuite-a-modular-simulation-framework-and","title":"robosuite: A Modular Simulation Framework and Benchmark for Robot Learning","date":"2020-09-25","arxiv_id":"2009.12293","repositories_listed":7,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/generating-holistic-3d-human-motion-from","title":"Generating Holistic 3D Human Motion from Speech","date":"2022-12-08","arxiv_id":"2212.04420","repositories_listed":3,"syntology":null},{"url":"/paper/the-genea-challenge-2022-a-large-evaluation","title":"The GENEA Challenge 2022: A large evaluation of data-driven co-speech gesture generation","date":"2022-08-22","arxiv_id":"2208.10441","repositories_listed":3,"syntology":null},{"url":"/paper/the-genea-challenge-2023-a-large-scale","title":"The GENEA Challenge 2023: A large scale evaluation of gesture generation models in monadic and dyadic settings","date":"2023-08-24","arxiv_id":"2308.12646","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/speech-gesture-generation-from-the-trimodal","title":"Speech Gesture Generation from the Trimodal Context of Text, Audio, and Speaker Identity","date":"2020-09-04","arxiv_id":"2009.02119","repositories_listed":2,"syntology":null},{"url":"/paper/learning-individual-styles-of-conversational-1","title":"Learning Individual Styles of Conversational Gesture","date":"2019-06-10","arxiv_id":"1906.04160","repositories_listed":2,"syntology":null},{"url":"/paper/deepgesture-a-conversational-gesture","title":"DeepGesture: A conversational gesture synthesis system based on emotions and semantics","date":"2025-07-03","arxiv_id":"2507.03147","repositories_listed":1,"syntology":null},{"url":"/paper/intentional-gesture-deliver-your-intentions","title":"Intentional Gesture: Deliver Your Intentions with Gestures for Speech","date":"2025-05-21","arxiv_id":"2505.15197","repositories_listed":1,"syntology":null},{"url":"/paper/varges-improving-variation-in-co-speech-3d","title":"VarGes: Improving Variation in Co-Speech 3D Gesture Generation via StyleCLIPS","date":"2025-02-15","arxiv_id":"2502.10729","repositories_listed":1,"syntology":null},{"url":"/paper/gesturelsm-latent-shortcut-based-co-speech","title":"GestureLSM: Latent Shortcut based Co-Speech Gesture Generation with Spatial-Temporal Modeling","date":"2025-01-31","arxiv_id":"2501.18898","repositories_listed":1,"syntology":null},{"url":"/paper/retrieving-semantics-from-the-deep-an-rag","title":"Retrieving Semantics from the Deep: an RAG Solution for Gesture Synthesis","date":"2024-12-09","arxiv_id":"2412.06786","repositories_listed":1,"syntology":null},{"url":"/paper/enabling-synergistic-full-body-control-in-1","title":"Enabling Synergistic Full-Body Control in Prompt-Based Co-Speech Motion Generation","date":"2024-10-01","arxiv_id":"2410.00464","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21136","title":"MotionCraft: Crafting Whole-Body Motion with Plug-and-Play Multimodal Controls","date":"2024-07-30","arxiv_id":"2407.21136","repositories_listed":1,"syntology":null},{"url":"/paper/amuse-emotional-speech-driven-3d-body","title":"AMUSE: Emotional Speech-driven 3D Body Animation via Disentangled Latent Diffusion","date":"2024-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/convofusion-multi-modal-conversational","title":"ConvoFusion: Multi-Modal Conversational Diffusion for Co-Speech Gesture Synthesis","date":"2024-03-26","arxiv_id":"2403.17936","repositories_listed":1,"syntology":null},{"url":"/paper/mambatalk-efficient-holistic-gesture","title":"MambaTalk: Efficient Holistic Gesture Synthesis with Selective State Space Models","date":"2024-03-14","arxiv_id":"2403.09471","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/emage-towards-unified-holistic-co-speech","title":"EMAGE: Towards Unified Holistic Co-Speech Gesture Generation via Expressive Masked Audio Gesture Modeling","date":"2023-12-31","arxiv_id":"2401.00374","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":4}},{"url":"/paper/emotional-speech-driven-3d-body-animation-via","title":"Emotional Speech-driven 3D Body Animation via Disentangled Latent Diffusion","date":"2023-12-07","arxiv_id":"2312.04466","repositories_listed":1,"syntology":null},{"url":"/paper/livelyspeaker-towards-semantic-aware-co","title":"LivelySpeaker: Towards Semantic-Aware Co-Speech Gesture Generation","date":"2023-09-17","arxiv_id":"2309.09294","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_unverified":2,"n_pointer_only":10}},{"url":"/paper/unifiedgesture-a-unified-gesture-synthesis","title":"UnifiedGesture: A Unified Gesture Synthesis Model for Multiple Skeletons","date":"2023-09-13","arxiv_id":"2309.07051","repositories_listed":1,"syntology":null},{"url":"/paper/synthogestures-a-novel-framework-for","title":"SynthoGestures: A Novel Framework for Synthetic Dynamic Hand Gesture Generation for Driving Scenarios","date":"2023-09-08","arxiv_id":"2309.04421","repositories_listed":1,"syntology":null},{"url":"/paper/c2g2-controllable-co-speech-gesture","title":"C2G2: Controllable Co-speech Gesture Generation with Latent Diffusion Model","date":"2023-08-29","arxiv_id":"2308.15016","repositories_listed":1,"syntology":null},{"url":"/paper/emotiongesture-audio-driven-diverse-emotional","title":"EmotionGesture: Audio-Driven Diverse Emotional Co-Speech 3D Gesture Generation","date":"2023-05-30","arxiv_id":"2305.18891","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/qpgesture-quantization-based-and-phase-guided-1","title":"QPGesture: Quantization-Based and Phase-Guided Motion Matching for Natural Speech-Driven Gesture Generation","date":"2023-05-18","arxiv_id":"2305.11094","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/aq-gt-a-temporally-aligned-and-quantized-gru","title":"AQ-GT: a Temporally Aligned and Quantized GRU-Transformer for Co-Speech Gesture Synthesis","date":"2023-05-02","arxiv_id":"2305.01241","repositories_listed":1,"syntology":null},{"url":"/paper/gesturediffuclip-gesture-diffusion-model-with","title":"GestureDiffuCLIP: Gesture Diffusion Model with CLIP Latents","date":"2023-03-26","arxiv_id":"2303.14613","repositories_listed":1,"syntology":null},{"url":"/paper/taming-diffusion-models-for-audio-driven-co","title":"Taming Diffusion Models for Audio-Driven Co-Speech Gesture Generation","date":"2023-03-16","arxiv_id":"2303.09119","repositories_listed":1,"syntology":null},{"url":"/paper/listen-denoise-action-audio-driven-motion","title":"Listen, Denoise, Action! Audio-Driven Motion Synthesis with Diffusion Models","date":"2022-11-17","arxiv_id":"2211.09707","repositories_listed":1,"syntology":null},{"url":"/paper/rhythmic-gesticulator-rhythm-aware-co-speech","title":"Rhythmic Gesticulator: Rhythm-Aware Co-Speech Gesture Synthesis with Hierarchical Neural Embeddings","date":"2022-10-04","arxiv_id":"2210.01448","repositories_listed":1,"syntology":null},{"url":"/paper/zeroeggs-zero-shot-example-based-gesture","title":"ZeroEGGS: Zero-shot Example-based Gesture Generation from Speech","date":"2022-09-15","arxiv_id":"2209.07556","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}