{"url":"/dataset/ochuman","name":"OCHuman","full_name":null,"description_markdown":"This dataset focuses on heavily occluded human with comprehensive annotations including bounding-box, humans pose and instance mask. This dataset contains 13,360 elaborately annotated human instances within 5081 images. With average 0.573 MaxIoU of each person, **OCHuman** is the most complex and challenging dataset related to human.\n\nSource: [https://github.com/liruilong940607/OCHumanApi](https://github.com/liruilong940607/OCHumanApi)\nImage Source: [https://github.com/liruilong940607/OCHumanApi](https://github.com/liruilong940607/OCHumanApi)","description_withheld":null,"homepage":"https://github.com/liruilong940607/OCHumanApi","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/pose2seg-detection-free-human-instance","title":"Pose2Seg: Detection Free Human Instance Segmentation","first_author":"Song-Hai Zhang","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Pose Estimation","url":"/task/pose-estimation","datasets_with_task":"/datasets/task/pose-estimation"},{"name":"2D Human Pose Estimation","url":"/task/2d-human-pose-estimation","datasets_with_task":"/datasets/task/2d-human-pose-estimation"},{"name":"Multi-Person Pose Estimation","url":"/task/multi-person-pose-estimation","datasets_with_task":"/datasets/task/multi-person-pose-estimation"},{"name":"Keypoint Detection","url":"/task/keypoint-detection","datasets_with_task":"/datasets/task/keypoint-detection"},{"name":"Human Instance Segmentation","url":"/task/human-instance-segmentation","datasets_with_task":"/datasets/task/human-instance-segmentation"},{"name":"Pose-Based Human Instance Segmentation","url":"/task/pose-based-human-instance-segmentation","datasets_with_task":"/datasets/task/pose-based-human-instance-segmentation"}],"languages":[],"variants":["OCHuman"],"data_loaders":[{"repo":"https://github.com/open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose/blob/master/docs/tasks/2d_body_keypoint.md#ochuman","frameworks":["pytorch"]},{"repo":"https://github.com/liruilong940607/OCHumanApi","url":"https://github.com/liruilong940607/OCHumanApi","frameworks":[]}],"num_papers_in_archive":66,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/pose-estimation-on-ochuman","task":"Pose Estimation","dataset_variant":"OCHuman","rows":19,"metrics":["Test AP","Validation AP"],"first_row_in_archive_order":{"model":"ViTPose (ViTAE-G, GT bounding boxes)","paper":"/paper/vitpose-simple-vision-transformer-baselines","metrics":{"Test AP":"93.3","Validation AP":"92.8"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"vitae-transformer/vitpose","url":"https://github.com/vitae-transformer/vitpose"},{"title":"vitae-transformer/qformer","url":"https://github.com/vitae-transformer/qformer"},{"title":"JunkyByte/easy_ViTPose","url":"https://github.com/JunkyByte/easy_ViTPose"},{"title":"jaehyunnn/ViTPose_pytorch","url":"https://github.com/jaehyunnn/ViTPose_pytorch"},{"title":"gpastal24/ViTPose-Pytorch","url":"https://github.com/gpastal24/ViTPose-Pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/human-instance-segmentation-on-ochuman","task":"Human Instance Segmentation","dataset_variant":"OCHuman","rows":18,"metrics":["AP"],"first_row_in_archive_order":{"model":"BBox-Mask-Pose 2x","paper":"/paper/detection-pose-estimation-and-segmentation-1","metrics":{"AP":"32.4"},"code_links":[{"title":"MiraPurkrabek/BBoxMaskPose","url":"https://github.com/MiraPurkrabek/BBoxMaskPose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/2d-human-pose-estimation-on-ochuman","task":"2D Human Pose Estimation","dataset_variant":"OCHuman","rows":11,"metrics":["Test AP","Validation AP"],"first_row_in_archive_order":{"model":"BBox-Mask-Pose 2x","paper":"/paper/detection-pose-estimation-and-segmentation-1","metrics":{"Test AP":"48.3","Validation AP":"48.6"},"code_links":[{"title":"MiraPurkrabek/BBoxMaskPose","url":"https://github.com/MiraPurkrabek/BBoxMaskPose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/keypoint-detection-on-ochuman","task":"Keypoint Detection","dataset_variant":"OCHuman","rows":10,"metrics":["Test AP","Validation AP"],"first_row_in_archive_order":{"model":"BBox-Mask-Pose 2x","paper":"/paper/detection-pose-estimation-and-segmentation-1","metrics":{"Test AP":"48.3","Validation AP":"48.6"},"code_links":[{"title":"MiraPurkrabek/BBoxMaskPose","url":"https://github.com/MiraPurkrabek/BBoxMaskPose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-person-pose-estimation-on-ochuman","task":"Multi-Person Pose Estimation","dataset_variant":"OCHuman","rows":8,"metrics":["AP50","AP75","Validation AP"],"first_row_in_archive_order":{"model":"MIPNet (gt-bb)","paper":"/paper/multi-hypothesis-pose-networks-rethinking-top","metrics":{"AP50":"89.7","AP75":"80.1","Validation AP":"74.1"},"code_links":[{"title":"rawalkhirodkar/MIPNet","url":"https://github.com/rawalkhirodkar/MIPNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-based-human-instance-segmentation-on","task":"Pose-Based Human Instance Segmentation","dataset_variant":"OCHuman","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"Pose2Seg (plus ground-truth keypoints)","paper":"/paper/pose2seg-detection-free-human-instance","metrics":{"AP":"55.2"},"code_links":[{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"liruilong940607/Pose2Seg","url":"https://github.com/liruilong940607/Pose2Seg"},{"title":"liruilong940607/OCHumanApi","url":"https://github.com/liruilong940607/OCHumanApi"},{"title":"Jittor/InstanceSegmentation-jittor","url":"https://github.com/Jittor/InstanceSegmentation-jittor"},{"title":"ligaoqi2/Pose2Seg-single-person-video-demo","url":"https://github.com/ligaoqi2/Pose2Seg-single-person-video-demo"},{"title":"jacksonlli/ProjectQuarantine_HumanSegmentation","url":"https://github.com/jacksonlli/ProjectQuarantine_HumanSegmentation"},{"title":"hz-ants/Pose2Seg","url":"https://github.com/hz-ants/Pose2Seg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/posebh-prototypical-multi-dataset-training","title":"PoseBH: Prototypical Multi-Dataset Training Beyond Human Pose Estimation","date":"2025-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/detection-pose-estimation-and-segmentation-1","title":"Detection, Pose Estimation and Segmentation for Multiple Bodies: Closing the Virtuous Circle","date":"2024-12-02","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/crowd-sam-sam-as-a-smart-annotator-for-object","title":"Crowd-SAM: SAM as a Smart Annotator for Object Detection in Crowded Scenes","date":"2024-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/you-only-learn-one-query-learning-unified","title":"You Only Learn One Query: Learning Unified Human Query for Single-Stage Multi-Person Multi-Task Human-Centric Perception","date":"2023-12-09","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-pose-estimation-in-crowds","title":"Rethinking pose estimation in crowds: overcoming the detection information-bottleneck and ambiguity","date":"2023-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rtmpose-real-time-multi-person-pose","title":"RTMPose: Real-Time Multi-Person Pose Estimation based on MMPose","date":"2023-03-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/object-centric-multi-task-learning-for-human","title":"Object-Centric Multi-Task Learning for Human Instances","date":"2023-03-13","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/unihcp-a-unified-model-for-human-centric","title":"UniHCP: A Unified Model for Human-Centric Perceptions","date":"2023-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":7,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sefd-learning-to-distill-complex-pose-and","title":"SEFD: Learning to Distill Complex Pose and Occlusion","date":"2023-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/test-time-adaptation-vs-training-time","title":"Test-time Adaptation vs. Training-time Generalization: A Case Study in Human Instance Segmentation using Keypoints Estimation","date":"2022-12-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/humans-need-not-label-more-humans-occlusion","title":"Humans need not label more humans: Occlusion Copy & Paste for Occluded Human Instance Segmentation","date":"2022-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/occlusion-aware-instance-segmentation-via","title":"Occlusion-Aware Instance Segmentation via BiLayer Network Architectures","date":"2022-08-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/i-2r-net-intra-and-inter-human-relation","title":"I^2R-Net: Intra- and Inter-Human Relation Network for Multi-Person Pose Estimation","date":"2022-06-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":18,"samples_unverified":13,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contextual-instance-decoupling-for-robust","title":"Contextual Instance Decoupling for Robust Multi-Person Pose Estimation","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hrformer-high-resolution-transformer-for","title":"HRFormer: High-Resolution Transformer for Dense Prediction","date":"2021-10-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/real-time-human-centric-segmentation-for","title":"Real-time Human-Centric Segmentation for Complex Video Scenes","date":"2021-08-16","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/multi-hypothesis-pose-networks-rethinking-top","title":"Multi-Instance Pose Networks: Rethinking Top-Down Pose Estimation","date":"2021-01-27","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transpose-towards-explainable-human-pose","title":"TransPose: Keypoint Localization via Transformer","date":"2020-12-28","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/count-and-similarity-aware-r-cnn-for","title":"Count- and Similarity-aware R-CNN for Pedestrian Detection","date":"2020-08-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/differentiable-hierarchical-graph-grouping","title":"Differentiable Hierarchical Graph Grouping for Multi-Person Pose Estimation","date":"2020-07-23","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/poseg-pose-aware-refinement-network-for-human","title":"PoSeg: Pose-Aware Refinement Network for Human Instance Segmentation","date":"2020-01-07","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/single-stage-multi-person-pose-machines","title":"Single-Stage Multi-Person Pose Machines","date":"2019-08-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/crowdpose-efficient-crowded-scenes-pose","title":"CrowdPose: Efficient Crowded Scenes Pose Estimation and A New Benchmark","date":"2018-12-02","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/simple-baselines-for-human-pose-estimation","title":"Simple Baselines for Human Pose Estimation and Tracking","date":"2018-04-17","rows_on_this_dataset":7,"code_links":27,"syntology":null},{"paper":"/paper/pose2seg-detection-free-human-instance","title":"Pose2Seg: Detection Free Human Instance Segmentation","date":"2018-03-28","rows_on_this_dataset":4,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":0,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mask-r-cnn","title":"Mask R-CNN","date":"2017-03-20","rows_on_this_dataset":1,"code_links":179,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":140,"samples_ran":42,"samples_unverified":98,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rmpe-regional-multi-person-pose-estimation","title":"RMPE: Regional Multi-person Pose Estimation","date":"2016-12-01","rows_on_this_dataset":3,"code_links":14,"syntology":null},{"paper":"/paper/associative-embedding-end-to-end-learning-for","title":"Associative Embedding: End-to-End Learning for Joint Detection and Grouping","date":"2016-11-16","rows_on_this_dataset":6,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":9,"samples_harvested":230,"samples_ran":79,"samples_unverified":151,"pointer_only_for_licence":29,"papers_with_no_sample_that_ran":3,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}