{"url":"/task/pose-tracking","name":"Pose Tracking","slug":"pose-tracking","description_markdown":"**Pose Tracking** is the task of estimating multi-person human poses in videos and assigning unique instance IDs for each keypoint across frames. Accurate estimation of human keypoint-trajectories is useful for human action recognition, human interaction understanding, motion capture and animation.\r\n\r\n\r\n<span class=\"description-source\">Source: [LightTrack: A Generic Framework for Online Top-Down Human Pose Tracking ](https://arxiv.org/abs/1905.02822)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":191,"papers_with_code":76,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":14,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/pose-tracking-on-posetrack2017","slug":"pose-tracking-on-posetrack2017","dataset":"PoseTrack2017","dataset_url":"/dataset/posetrack","rows_in_archive":10,"metrics":["MOTA","mAP"],"first_row_in_archive_order":{"model":"DetTrack","paper_title":"Combining detection and tracking for human pose estimation in videos","paper_url":"/paper/combining-detection-and-tracking-for-human","paper_date":"2020-03-30","arxiv_id":"2003.13743","code_links":[],"syntology":null}},{"leaderboard":"/sota/pose-tracking-on-posetrack2018","slug":"pose-tracking-on-posetrack2018","dataset":"PoseTrack2018","dataset_url":"/dataset/posetrack","rows_in_archive":5,"metrics":["MOTA","mAP","IDF1","IDs"],"first_row_in_archive_order":{"model":"DetTrack","paper_title":"Combining detection and tracking for human pose estimation in videos","paper_url":"/paper/combining-detection-and-tracking-for-human","paper_date":"2020-03-30","arxiv_id":"2003.13743","code_links":[],"syntology":null}},{"leaderboard":"/sota/pose-tracking-on-multi-person-posetrack","slug":"pose-tracking-on-multi-person-posetrack","dataset":"Multi-Person PoseTrack","dataset_url":null,"rows_in_archive":1,"metrics":["MOTA","MOTP"],"first_row_in_archive_order":{"model":"PoseTrack","paper_title":"PoseTrack: Joint Multi-Person Pose Estimation and Tracking","paper_url":"/paper/posetrack-joint-multi-person-pose-estimation","paper_date":"2016-11-23","arxiv_id":"1611.07727","code_links":[{"title":"umariqb/posetrack-cvpr2017","url":"https://github.com/umariqb/posetrack-cvpr2017"},{"title":"iqbalu/PoseTrack-CVPR2017","url":"https://github.com/iqbalu/PoseTrack-CVPR2017"}],"syntology":null}}],"datasets":[{"url":"/dataset/10000-people-human-pose-recognition-data","name":"10,000 People - Human Pose Recognition Data","full_name":"10,000 People - Human Pose Recognition Data","num_papers_in_archive":265},{"url":"/dataset/posetrack","name":"PoseTrack","full_name":"","num_papers_in_archive":103},{"url":"/dataset/hoi4d","name":"HOI4D","full_name":"","num_papers_in_archive":23},{"url":"/dataset/ycbineoat-dataset","name":"YCBInEOAT Dataset","full_name":"","num_papers_in_archive":6},{"url":"/dataset/vbr","name":"VBR","full_name":"VBR: A Vision Benchmark in Rome","num_papers_in_archive":3},{"url":"/dataset/drunkard-s-dataset","name":"Drunkard's Dataset","full_name":"","num_papers_in_archive":2},{"url":"/dataset/slam2ref","name":"SLAM2REF","full_name":"ConSLAM BIM and GT Poses","num_papers_in_archive":2},{"url":"/dataset/vrmocap-vr-mocap-dataset-for-pose","name":"VRMocap: VR Mocap Dataset for Pose Reconstruction","full_name":"","num_papers_in_archive":2},{"url":"/dataset/conslam","name":"ConSLAM","full_name":"Construction Dataset for SLAM","num_papers_in_archive":1},{"url":"/dataset/cope-119","name":"COPE-119","full_name":"","num_papers_in_archive":1},{"url":"/dataset/occluded-posetrack-reid","name":"Occluded-PoseTrack-ReID","full_name":"Occluded-PoseTrack Re-Identification","num_papers_in_archive":1},{"url":"/dataset/vr-mocap","name":"VR Mocap Dataset for Pose/Orientation Prediction","full_name":"","num_papers_in_archive":1},{"url":"/dataset/vr-folding","name":"VR-Folding","full_name":"","num_papers_in_archive":1},{"url":"/dataset/infiniterep","name":"InfiniteRep","full_name":"InfiniteRep","num_papers_in_archive":0}],"subtasks":[{"url":"/task/3d-human-pose-tracking","name":"3D Human Pose Tracking"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":76,"tagged_in_all":191,"items":[{"url":"/paper/deep-high-resolution-representation-learning","title":"Deep High-Resolution Representation Learning for Human Pose Estimation","date":"2019-02-25","arxiv_id":"1902.09212","repositories_listed":39,"syntology":{"n":25,"n_ran":8,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/simple-baselines-for-human-pose-estimation","title":"Simple Baselines for Human Pose Estimation and Tracking","date":"2018-04-17","arxiv_id":"1804.06208","repositories_listed":27,"syntology":null},{"url":"/paper/blazepose-on-device-real-time-body-pose","title":"BlazePose: On-device Real-time Body Pose tracking","date":"2020-06-17","arxiv_id":"2006.10204","repositories_listed":7,"syntology":null},{"url":"/paper/keypoint-promptable-re-identification","title":"Keypoint Promptable Re-Identification","date":"2024-07-25","arxiv_id":"2407.18112","repositories_listed":2,"syntology":null},{"url":"/paper/imu-aided-event-based-stereo-visual-odometry","title":"IMU-Aided Event-based Stereo Visual Odometry","date":"2024-05-07","arxiv_id":"2405.04071","repositories_listed":2,"syntology":null},{"url":"/paper/you-only-demonstrate-once-category-level","title":"You Only Demonstrate Once: Category-Level Manipulation from Single Visual Demonstration","date":"2022-01-30","arxiv_id":"2201.12716","repositories_listed":2,"syntology":{"n":30,"n_ran":4,"n_unverified":26,"n_pointer_only":0}},{"url":"/paper/roft-real-time-optical-flow-aided-6d-object","title":"ROFT: Real-Time Optical Flow-Aided 6D Object Pose and Velocity Tracking","date":"2021-11-06","arxiv_id":"2111.03821","repositories_listed":2,"syntology":null},{"url":"/paper/srt3d-a-sparse-region-based-3d-object","title":"SRT3D: A Sparse Region-Based 3D Object Tracking Approach for the Real World","date":"2021-10-25","arxiv_id":"2110.12715","repositories_listed":2,"syntology":null},{"url":"/paper/pose-id-on-a-novel-framework-for-artwork-pose","title":"POSE-ID-on—A Novel Framework for Artwork Pose Clustering","date":"2021-04-11","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/6-pack-category-level-6d-pose-tracker-with","title":"6-PACK: Category-level 6D Pose Tracker with Anchor-Based Keypoints","date":"2019-10-23","arxiv_id":"1910.10750","repositories_listed":2,"syntology":{"n":15,"n_ran":1,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/lighttrack-a-generic-framework-for-online-top","title":"LightTrack: A Generic Framework for Online Top-Down Human Pose Tracking","date":"2019-05-07","arxiv_id":"1905.02822","repositories_listed":2,"syntology":{"n":16,"n_ran":2,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/multigrid-predictive-filter-flow-for","title":"Multigrid Predictive Filter Flow for Unsupervised Learning on Videos","date":"2019-04-02","arxiv_id":"1904.01693","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/posetrack-a-benchmark-for-human-pose","title":"PoseTrack: A Benchmark for Human Pose Estimation and Tracking","date":"2017-10-27","arxiv_id":"1710.10000","repositories_listed":2,"syntology":null},{"url":"/paper/capturing-hand-motion-with-an-rgb-d-sensor","title":"Capturing Hand Motion with an RGB-D Sensor, Fusing a Generative Model with Salient Points","date":"2017-04-03","arxiv_id":"1704.00515","repositories_listed":2,"syntology":null},{"url":"/paper/posetrack-joint-multi-person-pose-estimation","title":"PoseTrack: Joint Multi-Person Pose Estimation and Tracking","date":"2016-11-23","arxiv_id":"1611.07727","repositories_listed":2,"syntology":null},{"url":"/paper/event-based-camera-pose-tracking-using-a","title":"Event-based Camera Pose Tracking using a Generative Event Model","date":"2015-10-07","arxiv_id":"1510.01972","repositories_listed":2,"syntology":null},{"url":"/paper/rgbtrack-fast-robust-depth-free-6d-pose","title":"RGBTrack: Fast, Robust Depth-Free 6D Pose Estimation and Tracking","date":"2025-06-20","arxiv_id":"2506.17119","repositories_listed":1,"syntology":null},{"url":"/paper/spikevideoformer-an-efficient-spike-driven","title":"SpikeVideoFormer: An Efficient Spike-Driven Video Transformer with Hamming Attention and $\\mathcal{O}(T)$ Complexity","date":"2025-05-15","arxiv_id":"2505.10352","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/embracing-dynamics-dynamics-aware-4d-gaussian","title":"Embracing Dynamics: Dynamics-aware 4D Gaussian Splatting SLAM","date":"2025-04-07","arxiv_id":"2504.04844","repositories_listed":1,"syntology":null},{"url":"/paper/dynopets-a-versatile-benchmark-for-dynamic","title":"DynOPETs: A Versatile Benchmark for Dynamic Object Pose Estimation and Tracking in Moving Camera Scenarios","date":"2025-03-25","arxiv_id":"2503.19625","repositories_listed":1,"syntology":null},{"url":"/paper/stereo-event-based-6-dof-pose-tracking-for","title":"Stereo Event-based, 6-DOF Pose Tracking for Uncooperative Spacecraft","date":"2025-03-17","arxiv_id":"2503.12732","repositories_listed":1,"syntology":null},{"url":"/paper/g3flow-generative-3d-semantic-flow-for-pose","title":"G3Flow: Generative 3D Semantic Flow for Pose-aware and Generalizable Object Manipulation","date":"2024-11-27","arxiv_id":"2411.18369","repositories_listed":1,"syntology":null},{"url":"/paper/segment-anything-in-light-fields-for-real","title":"Segment Anything in Light Fields for Real-Time Applications via Constrained Prompting","date":"2024-11-21","arxiv_id":"2411.13840","repositories_listed":1,"syntology":null},{"url":"/paper/esvo2-direct-visual-inertial-odometry-with","title":"ESVO2: Direct Visual-Inertial Odometry with Stereo Event Cameras","date":"2024-10-12","arxiv_id":"2410.09374","repositories_listed":1,"syntology":null},{"url":"/paper/gs-evt-cross-modal-event-camera-tracking","title":"GS-EVT: Cross-Modal Event Camera Tracking based on Gaussian Splatting","date":"2024-09-28","arxiv_id":"2409.19228","repositories_listed":1,"syntology":null},{"url":"/paper/srpose-two-view-relative-pose-estimation-with","title":"SRPose: Two-view Relative Pose Estimation with Sparse Keypoints","date":"2024-07-11","arxiv_id":"2407.08199","repositories_listed":1,"syntology":null},{"url":"/paper/ho-cap-a-capture-system-and-dataset-for-3d","title":"HO-Cap: A Capture System and Dataset for 3D Reconstruction and Pose Tracking of Hand-Object Interaction","date":"2024-06-10","arxiv_id":"2406.06843","repositories_listed":1,"syntology":null},{"url":"/paper/high-fidelity-slam-using-gaussian-splatting","title":"High-Fidelity SLAM Using Gaussian Splatting with Rendering-Guided Densification and Regularized Optimization","date":"2024-03-19","arxiv_id":"2403.12535","repositories_listed":1,"syntology":null},{"url":"/paper/videomac-video-masked-autoencoders-meet","title":"VideoMAC: Video Masked Autoencoders Meet ConvNets","date":"2024-02-29","arxiv_id":"2402.19082","repositories_listed":1,"syntology":null},{"url":"/paper/towards-real-world-aerial-vision-guidance","title":"Towards Real-World Aerial Vision Guidance with Categorical 6D Pose Tracker","date":"2024-01-09","arxiv_id":"2401.04377","repositories_listed":1,"syntology":null}],"syntology_records":6,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}