{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/learning-individual-styles-of-conversational-1","title":"Learning Individual Styles of Conversational Gesture","arxiv_id":"1906.04160","date":"2019-06-10","proceeding":"CVPR 2019 6","authors":["Shiry Ginosar","Amir Bar","Gefen Kohavi","Caroline Chan","Andrew Owens","Jitendra Malik"],"abstract":"Human speech is often accompanied by hand and arm gestures. Given audio speech input, we generate plausible gestures to go along with the sound. Specifically, we perform cross-modal translation from \"in-the-wild'' monologue speech of a single speaker to their hand and arm motion. We train on unlabeled videos for which we only have noisy pseudo ground truth from an automatic pose detection system. Our proposed model significantly outperforms baseline methods in a quantitative comparison. To support research toward obtaining a computational understanding of the relationship between gesture and speech, we release a large video dataset of person-specific gestures. The project website with video, code and data can be found at http://people.eecs.berkeley.edu/~shiry/speech2gesture .","url_abs":"https://arxiv.org/abs/1906.04160v1","url_pdf":"https://arxiv.org/pdf/1906.04160v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"learning-individual-styles-of-conversational-1","repo_url":"https://github.com/amirbar/speech2gesture","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":{"status":"ok"}},{"paper_slug":"learning-individual-styles-of-conversational-1","repo_url":"https://github.com/PantoMatrix/BEAT","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":"gesture-generation","task_name":"Gesture Generation"},{"task_slug":"speech-to-gesture-translation","task_name":"Speech-to-Gesture Translation"},{"task_slug":"translation","task_name":"Translation"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/gesture-generation-on-beat","task":"Gesture Generation","dataset":"BEAT","model":"Speech2Gestures","rank_in_archive_order":4,"of":5,"metrics":{"FID":"256.7"},"uses_additional_data":false},{"leaderboard":"/sota/gesture-generation-on-beat2","task":"Gesture Generation","dataset":"BEAT2","model":"S2G","rank_in_archive_order":14,"of":14,"metrics":{"FGD":"2.815"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/1906.04160","atlas_url":"https://app.syntology.ai/?focus=1906.04160","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}