{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/unipose-unified-human-pose-estimation-in","title":"UniPose: Unified Human Pose Estimation in Single Images and Videos","arxiv_id":"2001.08095","date":"2020-01-22","proceeding":"CVPR 2020 6","authors":["Bruno Artacho","Andreas Savakis"],"abstract":"We propose UniPose, a unified framework for human pose estimation, based on our \"Waterfall\" Atrous Spatial Pooling architecture, that achieves state-of-art-results on several pose estimation metrics. Current pose estimation methods utilizing standard CNN architectures heavily rely on statistical postprocessing or predefined anchor poses for joint localization. UniPose incorporates contextual segmentation and joint localization to estimate the human pose in a single stage, with high accuracy, without relying on statistical postprocessing methods. The Waterfall module in UniPose leverages the efficiency of progressive filtering in the cascade architecture, while maintaining multi-scale fields-of-view comparable to spatial pyramid configurations. Additionally, our method is extended to UniPose-LSTM for multi-frame processing and achieves state-of-the-art results for temporal pose estimation in Video. Our results on multiple datasets demonstrate that UniPose, with a ResNet backbone and Waterfall module, is a robust and efficient architecture for pose estimation obtaining state-of-the-art results in single person pose detection for both single images and videos.","url_abs":"https://arxiv.org/abs/2001.08095v1","url_pdf":"https://arxiv.org/pdf/2001.08095v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"unipose-unified-human-pose-estimation-in","repo_url":"https://github.com/bmartacho/UniPose","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"unipose-unified-human-pose-estimation-in","repo_url":"https://github.com/yangyucheng000/unipose-mindspore","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"mindspore","reach":null}],"tasks":[{"task_slug":"pose-estimation","task_name":"Pose Estimation"},{"task_slug":"skeleton-based-action-recognition","task_name":"Skeleton Based Action Recognition"}],"methods":[{"method_slug":"1x1-convolution","method_name":"1x1 Convolution"},{"method_slug":"average-pooling","method_name":"Average Pooling"},{"method_slug":"batch-normalization","method_name":"Batch Normalization"},{"method_slug":"bottleneck-residual-block","method_name":"Bottleneck Residual Block"},{"method_slug":"convolution","method_name":"Convolution"},{"method_slug":"global-average-pooling","method_name":"Global Average Pooling"},{"method_slug":"kaiming-initialization","method_name":"Kaiming Initialization"},{"method_slug":"max-pooling","method_name":"Max Pooling"},{"method_slug":"relu","method_name":"ReLU"},{"method_slug":"residual-block","method_name":"Residual Block"},{"method_slug":"residual-connection","method_name":"Residual Connection"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/pose-estimation-on-leeds-sports-poses","task":"Pose Estimation","dataset":"Leeds Sports Poses","model":"UniPose","rank_in_archive_order":3,"of":18,"metrics":{"PCK":"94.5%"},"uses_additional_data":false},{"leaderboard":"/sota/pose-estimation-on-mpii-human-pose","task":"Pose Estimation","dataset":"MPII Human Pose","model":"UniPose","rank_in_archive_order":8,"of":46,"metrics":{"PCKh-0.5":"92.7"},"uses_additional_data":false},{"leaderboard":"/sota/pose-estimation-on-upenn-action","task":"Pose Estimation","dataset":"UPenn Action","model":"UniPose-LSTM","rank_in_archive_order":2,"of":5,"metrics":{"Mean PCK@0.2":"99.3"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2001.08095","atlas_url":"https://app.syntology.ai/?focus=2001.08095","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}