{"url":"/dataset/harper","name":"HARPER","full_name":"Exploring 3D Human Pose Estimation and Forecasting from the Robot’s Perspective: The HARPER Dataset","description_markdown":"We introduce HARPER, a novel dataset for 3D body pose estimation and forecast in dyadic interactions between users and \\spot, the quadruped robot manufactured by Boston Dynamics. The key-novelty is the focus on the robot's perspective, i.e., on the data captured by the robot's sensors. These make 3D body pose analysis challenging because being close to the ground captures humans only partially. The scenario underlying HARPER includes 15 actions, of which 10 involve physical contact between the robot and users. The Corpus contains not only the recordings of the built-in stereo cameras of Spot, but also those of a 6-camera OptiTrack system (all recordings are synchronized). This leads to ground-truth skeletal representations with a precision lower than a millimeter. In addition, the Corpus includes reproducible benchmarks on 3D Human Pose Estimation, Human Pose Forecasting, and Collision Prediction, all based on publicly available baseline approaches. This enables future HARPER users to rigorously compare their results with those we provide in this work.\r\n\r\nSource: [Download dataset] (https://github.com/intelligolabs/HARPER)","description_withheld":null,"homepage":"https://github.com/intelligolabs/HARPER","introduced_date":"2024-03-21","introduced_date_note":null,"introduced_by":{"paper":"/paper/exploring-3d-human-pose-estimation-and","title":"Exploring 3D Human Pose Estimation and Forecasting from the Robot's Perspective: The HARPER Dataset","first_author":"Andrea Avogaro","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"},{"name":"3D","url":"/datasets/modality/3d"},{"name":"RGB-D","url":"/datasets/modality/rgb-d"}],"tasks":[{"name":"3D Human Pose Estimation","url":"/task/3d-human-pose-estimation","datasets_with_task":"/datasets/task/3d-human-pose-estimation"},{"name":"2D Pose Estimation","url":"/task/2d-pose-estimation","datasets_with_task":"/datasets/task/2d-pose-estimation"},{"name":"Human Pose Forecasting","url":"/task/human-pose-forecasting","datasets_with_task":"/datasets/task/human-pose-forecasting"},{"name":"3D Pose Estimation","url":"/task/3d-pose-estimation","datasets_with_task":"/datasets/task/3d-pose-estimation"},{"name":"Pose Retrieval","url":"/task/pose-retrieval","datasets_with_task":"/datasets/task/pose-retrieval"},{"name":"Collision Avoidance","url":"/task/collision-avoidance","datasets_with_task":"/datasets/task/collision-avoidance"}],"languages":[],"variants":["HARPER"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/human-pose-forecasting-on-harper","task":"Human Pose Forecasting","dataset_variant":"HARPER","rows":3,"metrics":["Average MPJPE (mm) @ 400ms","Average MPJPE (mm) @ 1000ms","Last Frame MPJPE (mm) @ 400ms","Last Frame MPJPE (mm) @ 1000ms"],"first_row_in_archive_order":{"model":"EqMotion","paper":"/paper/eqmotion-equivariant-multi-agent-motion","metrics":{"Average MPJPE (mm) @ 1000ms":"104","Average MPJPE (mm) @ 400ms":"41","Last Frame MPJPE (mm) @ 1000ms":"197","Last Frame MPJPE (mm) @ 400ms":"69"},"code_links":[{"title":"mediabrain-sjtu/eqmotion","url":"https://github.com/mediabrain-sjtu/eqmotion"},{"title":"pranav-chib/trajimpute","url":"https://github.com/pranav-chib/trajimpute"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/2d-pose-estimation-on-harper","task":"2D Pose Estimation","dataset_variant":"HARPER","rows":1,"metrics":["PCK"],"first_row_in_archive_order":{"model":"HRNet","paper":"/paper/deep-high-resolution-representation-learning","metrics":{"PCK":"86,8"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection"},{"title":"PaddlePaddle/PaddleDetection","url":"https://github.com/PaddlePaddle/PaddleDetection"},{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"leoxiaobin/deep-high-resolution-net.pytorch","url":"https://github.com/leoxiaobin/deep-high-resolution-net.pytorch"},{"title":"HRNet/HRNet-Semantic-Segmentation","url":"https://github.com/HRNet/HRNet-Semantic-Segmentation"},{"title":"osmr/imgclsmob","url":"https://github.com/osmr/imgclsmob"},{"title":"Microsoft/human-pose-estimation.pytorch","url":"https://github.com/Microsoft/human-pose-estimation.pytorch"},{"title":"HRNet/HRNet-Facial-Landmark-Detection","url":"https://github.com/HRNet/HRNet-Facial-Landmark-Detection"},{"title":"HRNet/HRNet-Image-Classification","url":"https://github.com/HRNet/HRNet-Image-Classification"},{"title":"HRNet/HRNet-Object-Detection","url":"https://github.com/HRNet/HRNet-Object-Detection"},{"title":"mindspore-lab/mindone","url":"https://github.com/mindspore-lab/mindone"},{"title":"leeyegy/SimDR","url":"https://github.com/leeyegy/SimDR"},{"title":"leeyegy/simcc","url":"https://github.com/leeyegy/simcc"},{"title":"mks0601/PoseFix_RELEASE","url":"https://github.com/mks0601/PoseFix_RELEASE"},{"title":"HRNet/HRNet-Human-Pose-Estimation","url":"https://github.com/HRNet/HRNet-Human-Pose-Estimation"},{"title":"strivebo/image_segmentation_dl","url":"https://github.com/strivebo/image_segmentation_dl"},{"title":"HRNet/HRNet-MaskRCNN-Benchmark","url":"https://github.com/HRNet/HRNet-MaskRCNN-Benchmark"},{"title":"CASIA-IVA-Lab/ISP-reID","url":"https://github.com/CASIA-IVA-Lab/ISP-reID"},{"title":"NVlabs/PAMTRI","url":"https://github.com/NVlabs/PAMTRI"},{"title":"k-miran/hear","url":"https://github.com/k-miran/hear"},{"title":"Vill-Lab/2022-TIP-HCGA","url":"https://github.com/Vill-Lab/2022-TIP-HCGA"},{"title":"d-shivam/Pose-estimation-based-action-recognition-for-help-Situation-Identification","url":"https://github.com/d-shivam/Pose-estimation-based-action-recognition-for-help-Situation-Identification"},{"title":"chuanqichen/deepcoaching","url":"https://github.com/chuanqichen/deepcoaching"},{"title":"Mary-xl/HRnet_Kaggle_iNat2019_FGVC","url":"https://github.com/Mary-xl/HRnet_Kaggle_iNat2019_FGVC"},{"title":"v1viswan/Domain_adaptation_in_HRNet","url":"https://github.com/v1viswan/Domain_adaptation_in_HRNet"},{"title":"ken724049/action-recognition","url":"https://github.com/ken724049/action-recognition"},{"title":"NU-LL/lighttrack-","url":"https://github.com/NU-LL/lighttrack-"},{"title":"thoughtmachines/Human-Pose-Estimation-using-HRNets","url":"https://github.com/thoughtmachines/Human-Pose-Estimation-using-HRNets"},{"title":"ducongju/HRNet","url":"https://github.com/ducongju/HRNet"},{"title":"thomasslloyd/FitSpatial","url":"https://github.com/thomasslloyd/FitSpatial"},{"title":"laowang666888/HRNET","url":"https://github.com/laowang666888/HRNET"},{"title":"baoshengyu/deep-high-resolution-net.pytorch","url":"https://github.com/baoshengyu/deep-high-resolution-net.pytorch"},{"title":"sdll/hrnet-pose-estimation","url":"https://github.com/sdll/hrnet-pose-estimation"},{"title":"gox-ai/hrnet-pose-api","url":"https://github.com/gox-ai/hrnet-pose-api"},{"title":"anshky/HR-NET","url":"https://github.com/anshky/HR-NET"},{"title":"wsjzha/deep-high-resolution-net.pytorch","url":"https://github.com/wsjzha/deep-high-resolution-net.pytorch"},{"title":"visionNoob/hrnet_pytorch","url":"https://github.com/visionNoob/hrnet_pytorch"},{"title":"abhi1kumar/hrnet_pose_single_gpu","url":"https://github.com/abhi1kumar/hrnet_pose_single_gpu"},{"title":"goutern/PoseEstimation","url":"https://github.com/goutern/PoseEstimation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-pose-estimation-on-harper","task":"3D Pose Estimation","dataset_variant":"HARPER","rows":1,"metrics":["Average MPJPE (mm)"],"first_row_in_archive_order":{"model":"HRNet + Depth","paper":"/paper/deep-high-resolution-representation-learning","metrics":{"Average MPJPE (mm)":"151"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection"},{"title":"PaddlePaddle/PaddleDetection","url":"https://github.com/PaddlePaddle/PaddleDetection"},{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"leoxiaobin/deep-high-resolution-net.pytorch","url":"https://github.com/leoxiaobin/deep-high-resolution-net.pytorch"},{"title":"HRNet/HRNet-Semantic-Segmentation","url":"https://github.com/HRNet/HRNet-Semantic-Segmentation"},{"title":"osmr/imgclsmob","url":"https://github.com/osmr/imgclsmob"},{"title":"Microsoft/human-pose-estimation.pytorch","url":"https://github.com/Microsoft/human-pose-estimation.pytorch"},{"title":"HRNet/HRNet-Facial-Landmark-Detection","url":"https://github.com/HRNet/HRNet-Facial-Landmark-Detection"},{"title":"HRNet/HRNet-Image-Classification","url":"https://github.com/HRNet/HRNet-Image-Classification"},{"title":"HRNet/HRNet-Object-Detection","url":"https://github.com/HRNet/HRNet-Object-Detection"},{"title":"mindspore-lab/mindone","url":"https://github.com/mindspore-lab/mindone"},{"title":"leeyegy/SimDR","url":"https://github.com/leeyegy/SimDR"},{"title":"leeyegy/simcc","url":"https://github.com/leeyegy/simcc"},{"title":"mks0601/PoseFix_RELEASE","url":"https://github.com/mks0601/PoseFix_RELEASE"},{"title":"HRNet/HRNet-Human-Pose-Estimation","url":"https://github.com/HRNet/HRNet-Human-Pose-Estimation"},{"title":"strivebo/image_segmentation_dl","url":"https://github.com/strivebo/image_segmentation_dl"},{"title":"HRNet/HRNet-MaskRCNN-Benchmark","url":"https://github.com/HRNet/HRNet-MaskRCNN-Benchmark"},{"title":"CASIA-IVA-Lab/ISP-reID","url":"https://github.com/CASIA-IVA-Lab/ISP-reID"},{"title":"NVlabs/PAMTRI","url":"https://github.com/NVlabs/PAMTRI"},{"title":"k-miran/hear","url":"https://github.com/k-miran/hear"},{"title":"Vill-Lab/2022-TIP-HCGA","url":"https://github.com/Vill-Lab/2022-TIP-HCGA"},{"title":"d-shivam/Pose-estimation-based-action-recognition-for-help-Situation-Identification","url":"https://github.com/d-shivam/Pose-estimation-based-action-recognition-for-help-Situation-Identification"},{"title":"chuanqichen/deepcoaching","url":"https://github.com/chuanqichen/deepcoaching"},{"title":"Mary-xl/HRnet_Kaggle_iNat2019_FGVC","url":"https://github.com/Mary-xl/HRnet_Kaggle_iNat2019_FGVC"},{"title":"v1viswan/Domain_adaptation_in_HRNet","url":"https://github.com/v1viswan/Domain_adaptation_in_HRNet"},{"title":"ken724049/action-recognition","url":"https://github.com/ken724049/action-recognition"},{"title":"NU-LL/lighttrack-","url":"https://github.com/NU-LL/lighttrack-"},{"title":"thoughtmachines/Human-Pose-Estimation-using-HRNets","url":"https://github.com/thoughtmachines/Human-Pose-Estimation-using-HRNets"},{"title":"ducongju/HRNet","url":"https://github.com/ducongju/HRNet"},{"title":"thomasslloyd/FitSpatial","url":"https://github.com/thomasslloyd/FitSpatial"},{"title":"laowang666888/HRNET","url":"https://github.com/laowang666888/HRNET"},{"title":"baoshengyu/deep-high-resolution-net.pytorch","url":"https://github.com/baoshengyu/deep-high-resolution-net.pytorch"},{"title":"sdll/hrnet-pose-estimation","url":"https://github.com/sdll/hrnet-pose-estimation"},{"title":"gox-ai/hrnet-pose-api","url":"https://github.com/gox-ai/hrnet-pose-api"},{"title":"anshky/HR-NET","url":"https://github.com/anshky/HR-NET"},{"title":"wsjzha/deep-high-resolution-net.pytorch","url":"https://github.com/wsjzha/deep-high-resolution-net.pytorch"},{"title":"visionNoob/hrnet_pytorch","url":"https://github.com/visionNoob/hrnet_pytorch"},{"title":"abhi1kumar/hrnet_pose_single_gpu","url":"https://github.com/abhi1kumar/hrnet_pose_single_gpu"},{"title":"goutern/PoseEstimation","url":"https://github.com/goutern/PoseEstimation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/eqmotion-equivariant-multi-agent-motion","title":"EqMotion: Equivariant Multi-agent Motion Prediction with Invariant Interaction Reasoning","date":"2023-03-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/back-to-mlp-a-simple-baseline-for-human","title":"Back to MLP: A Simple Baseline for Human Motion Prediction","date":"2022-07-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/space-time-separable-graph-convolutional-1","title":"Space-Time-Separable Graph Convolutional Network for Pose Forecasting","date":"2021-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-high-resolution-representation-learning","title":"Deep High-Resolution Representation Learning for Human Pose Estimation","date":"2019-02-25","rows_on_this_dataset":2,"code_links":39,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":8,"samples_unverified":17,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":37,"samples_ran":14,"samples_unverified":23,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}