{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/vm-bhinet-vision-mamba-bimanual-hand","title":"VM-BHINet:Vision Mamba Bimanual Hand Interaction Network for 3D Interacting Hand Mesh Recovery From a Single RGB Image","arxiv_id":"2504.14618","date":"2025-04-20","proceeding":null,"authors":["Han Bi","Ge Yu","Yu He","Wenzhuo LIU","Zijie Zheng"],"abstract":"Understanding bimanual hand interactions is essential for realistic 3D pose and shape reconstruction. However, existing methods struggle with occlusions, ambiguous appearances, and computational inefficiencies. To address these challenges, we propose Vision Mamba Bimanual Hand Interaction Network (VM-BHINet), introducing state space models (SSMs) into hand reconstruction to enhance interaction modeling while improving computational efficiency. The core component, Vision Mamba Interaction Feature Extraction Block (VM-IFEBlock), combines SSMs with local and global feature operations, enabling deep understanding of hand interactions. Experiments on the InterHand2.6M dataset show that VM-BHINet reduces Mean per-joint position error (MPJPE) and Mean per-vertex position error (MPVPE) by 2-3%, significantly surpassing state-of-the-art methods.","url_abs":"https://arxiv.org/abs/2504.14618v1","url_pdf":"https://arxiv.org/pdf/2504.14618v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[],"tasks":[{"task_slug":"3d-interacting-hand-pose-estimation","task_name":"3D Interacting Hand Pose Estimation"},{"task_slug":"computational-efficiency","task_name":"Computational Efficiency"},{"task_slug":"mamba","task_name":"Mamba"},{"task_slug":null,"task_name":"Position"},{"task_slug":"state-space-models","task_name":"State Space Models"}],"methods":[{"method_slug":"mamba","method_name":"Mamba"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/3d-interacting-hand-pose-estimation-on","task":"3D Interacting Hand Pose Estimation","dataset":"InterHand2.6M","model":"VM-BHINet","rank_in_archive_order":1,"of":9,"metrics":{"MPJPE Test":"5.09","MPVPE Test":"5.44","MRRPE Test":"-"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}