{"url":"/dataset/coco-wholebody","name":"COCO-WholeBody","full_name":null,"description_markdown":"**COCO-WholeBody** is an extension of [COCO](/dataset/coco) dataset with whole-body annotations. There are 4 types of bounding boxes (person box, face box, left-hand box, and right-hand box) and 133 keypoints (17 for body, 6 for feet, 68 for face and 42 for hands) annotations for each person in the image.\r\n\r\nSource: [Whole-Body Human Pose Estimation in the Wild](/paper/whole-body-human-pose-estimation-in-the-wild)\r\n\r\nImage source: [https://arxiv.org/pdf/2007.11858v1.pdf](https://arxiv.org/pdf/2007.11858v1.pdf)","description_withheld":null,"homepage":"https://github.com/jin-s13/COCO-WholeBody","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/whole-body-human-pose-estimation-in-the-wild","title":"Whole-Body Human Pose Estimation in the Wild","first_author":"Sheng Jin","url":null},"license":{"name":"CC-BY-NC 4.0  ( not for commercial purpose)","url":"https://github.com/jin-s13/COCO-WholeBody?tab=readme-ov-file#terms-of-use"},"modalities":[],"tasks":[{"name":"Pose Estimation","url":"/task/pose-estimation","datasets_with_task":"/datasets/task/pose-estimation"},{"name":"2D Human Pose Estimation","url":"/task/2d-human-pose-estimation","datasets_with_task":"/datasets/task/2d-human-pose-estimation"},{"name":"Hand Pose Estimation","url":"/task/hand-pose-estimation","datasets_with_task":"/datasets/task/hand-pose-estimation"},{"name":"Facial Landmark Detection","url":"/task/facial-landmark-detection","datasets_with_task":"/datasets/task/facial-landmark-detection"},{"name":"Face Detection","url":"/task/face-detection","datasets_with_task":"/datasets/task/face-detection"},{"name":"Multi-Person Pose Estimation","url":"/task/multi-person-pose-estimation","datasets_with_task":"/datasets/task/multi-person-pose-estimation"},{"name":"Foot keypoint detection","url":"/task/foot-keypoint-detection","datasets_with_task":"/datasets/task/foot-keypoint-detection"}],"languages":[],"variants":["COCO-WholeBody"],"data_loaders":[{"repo":"https://github.com/open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose/blob/master/docs/tasks/2d_wholebody_keypoint.md#coco-wholebody","frameworks":["pytorch"]},{"repo":"https://github.com/jin-s13/COCO-WholeBody","url":"https://github.com/jin-s13/COCO-WholeBody","frameworks":[]}],"num_papers_in_archive":33,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/2d-human-pose-estimation-on-coco-wholebody-1","task":"2D Human Pose Estimation","dataset_variant":"COCO-WholeBody","rows":15,"metrics":["WB","body","foot","face","hand"],"first_row_in_archive_order":{"model":"RTMW-x","paper":"/paper/rtmw-real-time-multi-person-2d-and-3d-whole","metrics":{"WB":"70.2","body":"76.3","face":"88.4","foot":"79.6","hand":"66.4"},"code_links":[{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/face-detection-on-coco-wholebody","task":"Face Detection","dataset_variant":"COCO-WholeBody","rows":2,"metrics":["AP","AP50","AP75","APL","APM"],"first_row_in_archive_order":{"model":"HPRNet (Hourglass-104)","paper":"/paper/hprnet-hierarchical-point-regression-for","metrics":{"AP":"56.4","AP50":"82.4","AP75":"67.1","APL":"63.3","APM":"43.4"},"code_links":[{"title":"nerminsamet/HPRNet","url":"https://github.com/nerminsamet/HPRNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/facial-landmark-detection-on-coco-wholebody","task":"Facial Landmark Detection","dataset_variant":"COCO-WholeBody","rows":2,"metrics":["keypoint AP"],"first_row_in_archive_order":{"model":"HPRNet (Hourglass-104)","paper":"/paper/hprnet-hierarchical-point-regression-for","metrics":{"keypoint AP":"75.4"},"code_links":[{"title":"nerminsamet/HPRNet","url":"https://github.com/nerminsamet/HPRNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/hand-pose-estimation-on-coco-wholebody","task":"Hand Pose Estimation","dataset_variant":"COCO-WholeBody","rows":2,"metrics":["keypoint AP"],"first_row_in_archive_order":{"model":"HPRNet (Hourglass-104)","paper":"/paper/hprnet-hierarchical-point-regression-for","metrics":{"keypoint AP":"50.4"},"code_links":[{"title":"nerminsamet/HPRNet","url":"https://github.com/nerminsamet/HPRNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-person-pose-estimation-on-coco-1","task":"Multi-Person Pose Estimation","dataset_variant":"COCO-WholeBody","rows":2,"metrics":["keypoint AP"],"first_row_in_archive_order":{"model":"HPRNet (Hourglass-104)","paper":"/paper/hprnet-hierarchical-point-regression-for","metrics":{"keypoint AP":"59.4"},"code_links":[{"title":"nerminsamet/HPRNet","url":"https://github.com/nerminsamet/HPRNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/pcnet-a-human-pose-compensation-network-based","title":"PCNet: a human pose compensation network based on incremental learning for sports actions estimation","date":"2024-11-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/sapiens-foundation-for-human-vision-models","title":"Sapiens: Foundation for Human Vision Models","date":"2024-08-22","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rtmw-real-time-multi-person-2d-and-3d-whole","title":"RTMW: Real-Time Multi-Person 2D and 3D Whole-body Pose Estimation","date":"2024-07-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/rtmpose-real-time-multi-person-pose","title":"RTMPose: Real-Time Multi-Person Pose Estimation based on MMPose","date":"2023-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vitpose-vision-transformer-foundation-model","title":"ViTPose++: Vision Transformer for Generic Body Pose Estimation","date":"2022-12-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zoomnas-searching-for-whole-body-human-pose","title":"ZoomNAS: Searching for Whole-body Human Pose Estimation in the Wild","date":"2022-08-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/not-all-tokens-are-equal-human-centric-visual","title":"Not All Tokens Are Equal: Human-centric Visual Analysis via Token Clustering Transformer","date":"2022-04-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/keypoint-communities","title":"Keypoint Communities","date":"2021-10-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hprnet-hierarchical-point-regression-for","title":"HPRNet: Hierarchical Point Regression for Whole-Body Human Pose Estimation","date":"2021-06-08","rows_on_this_dataset":9,"code_links":1,"syntology":null},{"paper":"/paper/whole-body-human-pose-estimation-in-the-wild","title":"Whole-Body Human Pose Estimation in the Wild","date":"2020-07-23","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/deep-high-resolution-representation-learning","title":"Deep High-Resolution Representation Learning for Human Pose Estimation","date":"2019-02-25","rows_on_this_dataset":1,"code_links":39,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":8,"samples_unverified":17,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/realtime-multi-person-2d-pose-estimation","title":"Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields","date":"2016-11-24","rows_on_this_dataset":1,"code_links":61,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":4,"samples_unverified":19,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/associative-embedding-end-to-end-learning-for","title":"Associative Embedding: End-to-End Learning for Joint Detection and Grouping","date":"2016-11-16","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":77,"samples_ran":13,"samples_unverified":64,"pointer_only_for_licence":4,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}