{"url":"/dataset/itop","name":"ITOP","full_name":"Invariant-Top View Dataset","description_markdown":"The **ITOP** dataset consists of 40K training and 10K testing depth images for each of the front-view and top-view tracks. This dataset contains depth images with 20 actors who perform 15 sequences each and is recorded by two Asus Xtion Pro cameras. The ground-truth of this dataset is the 3D coordinates of 15 body joints.\n\nSource: [V2V-PoseNet: Voxel-to-Voxel Prediction Network for Accurate 3D Hand and Human Pose Estimation from a Single Depth Map](https://arxiv.org/abs/1711.07399)\nImage Source: [https://www.youtube.com/watch?v=4gPI-GOf9wg](https://www.youtube.com/watch?v=4gPI-GOf9wg)","description_withheld":null,"homepage":"https://zenodo.org/record/3932973","introduced_date":"2016-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/towards-viewpoint-invariant-3d-human-pose","title":"Towards Viewpoint Invariant 3D Human Pose Estimation","first_author":"Albert Haque","url":null},"license":{"name":"CC BY 4.0","url":"https://creativecommons.org/licenses/by/4.0/legalcode"},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Pose Estimation","url":"/task/pose-estimation","datasets_with_task":"/datasets/task/pose-estimation"},{"name":"3D Human Pose Estimation","url":"/task/3d-human-pose-estimation","datasets_with_task":"/datasets/task/3d-human-pose-estimation"}],"languages":[],"variants":[" ITOP front-view","ITOP top-view","ITOP front-view","ITOP"],"data_loaders":[],"num_papers_in_archive":23,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/pose-estimation-on-itop-front-view","task":"Pose Estimation","dataset_variant":"ITOP front-view","rows":7,"metrics":["Mean mAP"],"first_row_in_archive_order":{"model":"AdaPose","paper":"/paper/sequential-3d-human-pose-estimation-using","metrics":{"Mean mAP":"93.38"},"code_links":[{"title":"Hmslab/Adapose","url":"https://github.com/Hmslab/Adapose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-estimation-on-itop-top-view","task":"Pose Estimation","dataset_variant":"ITOP top-view","rows":5,"metrics":["Mean mAP"],"first_row_in_archive_order":{"model":"DECA-D3","paper":"/paper/deca-deep-viewpoint-equivariant-human-pose","metrics":{"Mean mAP":"86.92"},"code_links":[{"title":"mmlab-cv/deca","url":"https://github.com/mmlab-cv/deca"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/3d-human-pose-estimation-on-itop-front-view-1","task":"3D Human Pose Estimation","dataset_variant":"ITOP front-view","rows":1,"metrics":["Mean mAP"],"first_row_in_archive_order":{"model":"SPiKE","paper":"/paper/spike-3d-human-pose-from-point-cloud","metrics":{"Mean mAP":"89.19"},"code_links":[{"title":"iballester/SPiKE","url":"https://github.com/iballester/SPiKE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/spike-3d-human-pose-from-point-cloud","title":"SPiKE: 3D Human Pose from Point Cloud Sequences","date":"2024-09-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/sequential-3d-human-pose-estimation-using","title":"Sequential 3D Human Pose Estimation Using Adaptive Point Cloud Sampling Strategy","date":"2021-08-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deca-deep-viewpoint-equivariant-human-pose","title":"DECA: Deep viewpoint-Equivariant human pose estimation using Capsule Autoencoders","date":"2021-08-19","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/a2j-anchor-to-joint-regression-network-for-3d","title":"A2J: Anchor-to-Joint Regression Network for 3D Articulated Pose Estimation from a Single Depth Image","date":"2019-08-27","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/v2v-posenet-voxel-to-voxel-prediction-network","title":"V2V-PoseNet: Voxel-to-Voxel Prediction Network for Accurate 3D Hand and Human Pose Estimation from a Single Depth Map","date":"2017-11-20","rows_on_this_dataset":2,"code_links":5,"syntology":null},{"paper":"/paper/towards-good-practices-for-deep-3d-hand-pose","title":"Towards Good Practices for Deep 3D Hand Pose Estimation","date":"2017-07-23","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/towards-viewpoint-invariant-3d-human-pose","title":"Towards Viewpoint Invariant 3D Human Pose Estimation","date":"2016-03-23","rows_on_this_dataset":2,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":9,"samples_ran":6,"samples_unverified":3,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}